Compare commits

..
198 Commits
Author SHA1 Message Date
Ryan Houdek e8127b92e8 Docs: Update for release FEX-2404 2024-04-05 15:20:24 -07:00
Ryan Houdek 7786c23405 Merge pull request #3556 from Sonicadvance1/move_app_config
FEXCore: Fixes priority of FEX_APP_CONFIG
2024-04-05 15:18:24 -07:00
Ryan Houdek 904646e93b FEXCore: Fixes priority of FEX_APP_CONFIG
This environment variable had an incorrect priority on the configuration
system. The expectation was higher priority than most other layers.

Now the only layer that has higher priority is the environment
variables.
2024-04-05 13:10:43 -07:00
Alyssa Rosenzweig c43af8e975 Merge pull request #3553 from alyssarosenzweig/ra/shifts-rework-easy
OpcodeDispatcher: clean up shifts
2024-04-05 11:36:33 -04:00
Alyssa Rosenzweig a787daae41 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:27 -04:00
Alyssa Rosenzweig a05cc06ab4 OpcodeDispatcher: unify imm/1-bit ASHR
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:15 -04:00
Alyssa Rosenzweig 031e756a78 OpcodeDispatcher: unify imm/1-bit SHR
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:15 -04:00
Alyssa Rosenzweig 2a9f1ce8cb OpcodeDispatcher: unify imm/1-bit SHL
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:15 -04:00
Alyssa Rosenzweig 8c53a9f051 OpcodeDispatcher: use LoadConstantShift for rotates
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:15 -04:00
Alyssa Rosenzweig cf26ec7898 OpcodeDispatcher: use LoadConstantShift for SHRD
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:15 -04:00
Alyssa Rosenzweig 582c3dae6e OpcodeDispatcher: use LoadConstantShift for SHLD
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:15 -04:00
Alyssa Rosenzweig 2abac03ab0 OpcodeDispatcher: add LoadConstantShift helper
shows up a bunch

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:15 -04:00
Alyssa Rosenzweig 8cc684fa12 OpcodeDispatcher: drop misinformed comment
tbnz only tests a single bit, not a mask.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:15 -04:00
Alyssa Rosenzweig d92de1d947 OpcodeDispatcher: drop result masking for shifts
flag calcs are fine with upper garbage.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-04 07:42:15 -04:00
Alyssa Rosenzweig 202a60b77a Merge pull request #3549 from alyssarosenzweig/constprop/dce
ConstProp: drop dead code
2024-04-03 11:22:30 -04:00
Alyssa Rosenzweig aa8d04c341 Merge pull request #3551 from alyssarosenzweig/opt/negate-adds
Negate more to inline constants
2024-04-03 11:22:01 -04:00
Alyssa Rosenzweig d6425d05f3 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-02 13:57:06 -04:00
Alyssa Rosenzweig e07c81a5e7 ConstProp: also negate sub -> add
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-02 13:56:59 -04:00
Alyssa Rosenzweig fa76961873 ConstProp: negate adds -> subs
the arm ops are equiv, even though the x86 isn't (due to inverted carry).

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-02 13:53:21 -04:00
Alyssa Rosenzweig b92c206db9 ConstProp: rm your deadcode
not sure who this is supposed to be helping.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-02 13:21:48 -04:00
Alyssa Rosenzweig efff942724 ConstProp: drop my deadcode
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-02 13:21:48 -04:00
Alyssa Rosenzweig 8d32113521 ConstProp: rm relic of x86 jit
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-02 13:21:48 -04:00
Alyssa Rosenzweig 37f2b417e4 Merge pull request #3546 from alyssarosenzweig/flag/cleanup
Minor cleanups around flags
2024-04-02 11:29:19 -04:00
Alyssa Rosenzweig bd0b5eceb8 Merge pull request #3545 from alyssarosenzweig/opt/pf-scalar
Use scalar integer code to calculate PF
2024-04-02 11:28:53 -04:00
Alyssa Rosenzweig b632f7215c Merge pull request #3544 from alyssarosenzweig/ra/zero-multiple
OpcodeDispatcher: drop ZeroMultipleFlags
2024-04-02 11:27:45 -04:00
Ryan Houdek e8abc88702 Merge pull request #3542 from alyssarosenzweig/ra/rep
Eliminate xblock liveness with rep cmp/lod/scas
2024-04-02 04:24:24 -07:00
Ryan Houdek 29c6281e11 Merge pull request #3539 from alyssarosenzweig/ra/rol-ror2
rewrite ROL/ROR
2024-04-02 00:17:08 -07:00
Ryan Houdek 4214d9bda0 Merge pull request #3538 from pmatos/OffsetofOoB
Fix reference to out of bounds address in offsetof
2024-04-01 19:41:57 -07:00
Alyssa Rosenzweig 0a7b3efb41 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 16:42:50 -04:00
Alyssa Rosenzweig 067a5444dc OpcodeDispatcher: use HandleNZ00Write
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 16:42:38 -04:00
Alyssa Rosenzweig c7f159972d OpcodeDispatcher: rm pointless NZCV loads
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 16:42:38 -04:00
Alyssa Rosenzweig a70d0a5dd4 OpcodeDispatcher: rm unnecessary NZCV dirtying
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 16:42:38 -04:00
Ryan Houdek cd9ffd2045 Merge pull request #3536 from alyssarosenzweig/ra/rcl-rcr
OpcodeDispatcher: eliminate xblock liveness for rcl/rcr
2024-04-01 11:44:37 -07:00
Alyssa Rosenzweig 5c590b9a50 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 14:13:09 -04:00
Alyssa Rosenzweig eb4bb5875e OpcodeDispatcher: absorb invert into PF calculation
with xorn

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 14:12:33 -04:00
Alyssa Rosenzweig 3b052e826f OpcodeDispatcher: calculate PF with integer ops
based on clang's __builtin_parity

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 14:12:32 -04:00
Alyssa Rosenzweig 65ec191dc1 IR: add XornShift
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 14:12:32 -04:00
Alyssa Rosenzweig b1ddd8cd3b Merge pull request #3541 from alyssarosenzweig/opt/clc
optimize clc
2024-04-01 13:51:10 -04:00
Alyssa Rosenzweig f2d001e721 Merge pull request #3543 from alyssarosenzweig/ra/dead-code
RA: drop dead block interference code
2024-04-01 13:51:00 -04:00
Alyssa Rosenzweig 7852909cc4 OpcodeDispatcher: simplify IsNZCV
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 13:50:00 -04:00
Alyssa Rosenzweig fad243d3f6 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 13:48:27 -04:00
Alyssa Rosenzweig f8b68d8b5a OpcodeDispatcher: drop ZeroMultipleFlags
lot of complexity for only a single interesting case. we can massively simplify.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-04-01 13:48:11 -04:00
Ryan Houdek e2a095372e Merge pull request #3534 from Sonicadvance1/move_ir_defines
FEXCore: Move nearly all IR definitions to internal
2024-04-01 10:00:20 -07:00
Ryan Houdek 5c29c9d464 Merge pull request #3527 from Sonicadvance1/move_type_defines
Moves FHU TypeDefines to FEXCore includes
2024-04-01 08:57:22 -07:00
Ryan Houdek 3bed305660 Merge pull request #3526 from Sonicadvance1/move_codeloader
FEXCore: Moves CodeLoader to frontend
2024-04-01 07:52:02 -07:00
Ryan Houdek f6639c3594 Merge pull request #3525 from Sonicadvance1/move_cpubackend
FEXCore: Moves CPUBackend definition internal
2024-04-01 06:47:34 -07:00
Paulo Matos 96087a69fa Fix reference to OoB address in offsetof and remove rflags printout
Adjust static array size to match new size.
Remove rflags from printing code and adjust offsets - fixes
printing off-by-one error.
2024-04-01 13:13:17 +02:00
Alyssa Rosenzweig ca1ec232c9 RA: drop dead block interference code
Unused, and new RA won't use it either. Torch it.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 20:51:11 -04:00
Alyssa Rosenzweig ad0dd34412 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 20:31:38 -04:00
Alyssa Rosenzweig 7b1bb159fa OpcodeDispatcher: use ForeachDirection for scas
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 20:31:38 -04:00
Alyssa Rosenzweig 5c7f2934de OpcodeDispatcher: use ForeachDirection for lods
eliminates xblock live

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 20:29:29 -04:00
Alyssa Rosenzweig 5d79d4eb50 OpcodeDispatcher: use ForeachDirection for CMPS
eliminates xblock liveness

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 20:29:16 -04:00
Alyssa Rosenzweig 3f66173bc7 OpcodeDispatcher: add ForeachDirection helper
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 20:28:56 -04:00
Alyssa Rosenzweig b64a594b16 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 20:01:44 -04:00
Alyssa Rosenzweig 4452f0acba ConstProp: optimize rmif with 0 for clc
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 20:01:44 -04:00
Alyssa Rosenzweig 784cdd7b6b InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 19:54:14 -04:00
Alyssa Rosenzweig 1a1545da0f OpcodeDispatcher: rework rep cmp
1. pull flag calculation out of the loop body for perf
2. fully rotate the inner loop to save an instruction per iteration
3. hoist the rcx=0 jump to avoid computing df when rcx=0

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 19:54:00 -04:00
Alyssa Rosenzweig a70ea30c02 IR: add CondSubNZCV (ccmp) instruction
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 17:50:57 -04:00
Alyssa Rosenzweig 2aa1fd7fa3 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 14:47:19 -04:00
Alyssa Rosenzweig 15b86e4c5a OpcodeDispatcher: rewrite ROL/ROR
single unified implementation for ROL & ROR (instead of 4 cases). no more
deferred flags because it's easy to shoot ourselves in the foot with deferred
flags w.r.t the new RA design, and rotates are rare enough with very efficient
flag calculations such that the extra JIT overhead should be minimal to DCE the
resulting calculations later.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-31 14:47:19 -04:00
Alyssa Rosenzweig 67baff8a57 Merge pull request #3537 from Sonicadvance1/remove_vla_ra
RA: Removes VLA usage
2024-03-31 14:44:37 -04:00
Ryan Houdek fedc24be1e RA: Removes VLA usage
Just like #3508, clang-18 complains about VLA usage.

This vector is relatively small, only around 18 elements but is
semi-dynamic depending on arch and if FEXCore is targeting Linux or
Win32.
2024-03-30 16:50:04 -07:00
Alyssa Rosenzweig 2a625a467b Merge pull request #3530 from alyssarosenzweig/opt/cmpxchg-flags2
Optimize cmpxchg with flagm
2024-03-30 15:22:40 -04:00
Alyssa Rosenzweig 6f5e4fd34b OpcodeDispatcher: add non-flag calc version of ShiftVariable
more correct for rcl, etc

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 14:30:36 -04:00
Alyssa Rosenzweig bdda99e44f ConstProp: constant fold Neg
will come up with rotate in the next patch

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 14:15:11 -04:00
Alyssa Rosenzweig 9bca052146 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 14:12:59 -04:00
Alyssa Rosenzweig 706065b0e2 OpcodeDispatcher: accelerate cmpxchg with flagm
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 14:12:59 -04:00
Alyssa Rosenzweig 9fd32f07cb JIT: preserve nzcv for the slow atomic path
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 14:12:59 -04:00
Alyssa Rosenzweig deba6a1b76 JIT: add comment about unaligned backpatching
save future me some grief.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 14:12:59 -04:00
Alyssa Rosenzweig 6866c3d0ac InstCountCI: add dead cmpxchg case
not super optimizable but worth tracking.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 14:12:59 -04:00
Alyssa Rosenzweig d25ace43aa Merge pull request #3528 from alyssarosenzweig/ra/xsave-xrstor
Eliminate crossblock liveness in xsave/xrstor
2024-03-30 14:11:25 -04:00
Alyssa Rosenzweig d3b2ddf641 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 13:42:41 -04:00
Alyssa Rosenzweig 9010b3c117 OpcodeDispatcher: use neg trick for rcl smaller
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 13:42:41 -04:00
Alyssa Rosenzweig b0e001b660 OpcodeDispatcher: elim xblock live for smaller rcl
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 13:42:41 -04:00
Alyssa Rosenzweig eadacbd67b OpcodeDispatcher: elim xblock live with smaller rcr
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 13:42:03 -04:00
Alyssa Rosenzweig 7de29749be OpcodeDispatcher: eliminate xblock live for rcl
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 13:42:03 -04:00
Alyssa Rosenzweig 1f3843ccad OpcodeDispatcher: eliminate xblock live for rcr
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 13:42:03 -04:00
Alyssa Rosenzweig bc76df9901 OpcodeDispatcher: add non-flag calc version of ShiftVariable
more correct for rcl, etc

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 13:42:03 -04:00
Alyssa Rosenzweig 6e92cc454d ConstProp: constant fold Neg
will come up with rotate in the next patch

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-30 13:42:03 -04:00
Ryan Houdek ed3af580c5 FEXCore: Move nearly all IR definitions to internal
It has been a long time coming that FEX no longer needed to leak IR
implementation details to the frontend, this was legacy due to IR CI and
various other problems.

Now that the last bits of IR leaking has been removed, move everything
that we can internally to the implementation.
We still have a couple of minor details in the exposed IR.h to the
frontend, but these are limited to a few enums and some thunking struct
information rather than all the implementation details.

No functional change with this, just moving headers around.
2024-03-29 17:20:18 -07:00
Mai 2ad170b7d7 Merge pull request #3533 from Sonicadvance1/remove_debugstore
FEXCore: Remove DebugStore map
2024-03-29 20:19:47 -04:00
Ryan Houdek 8564290f76 FEXCore: Remove DebugStore map
This hasn't been used and is blocking refactoring more code.
2024-03-29 14:58:44 -07:00
Alyssa Rosenzweig 1d96631af7 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-29 09:58:03 -04:00
Alyssa Rosenzweig c513b9685d OpcodeDispatcher: eliminate crossblock liveness in xsave/xrstor
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-29 09:57:16 -04:00
Ryan Houdek d11a36eaea Moves FHU TypeDefines to FEXCore includes
FEXCore includes was including an FHU header which would result in
compilation failure for external projects trying to link to libFEXCore.

Moves it over to fix this, it was the only FHU usage in FEXCore/include
NFC
2024-03-29 02:54:54 -07:00
Ryan Houdek f46e88ebdb FEXCore: Moves CPUBackend definition internal
This is no longer necessary to be part of the public API. Moves the
header internally.

Needed to pass through `IsAddressInCodeBuffer` from CPUBackend through
the Context object, but otherwise no functional change.
2024-03-29 02:27:29 -07:00
Ryan Houdek 20eb338644 FEXCore: Moves CodeLoader to frontend
FEXCore no longer has a need for this since a bunch of related code was
already moved to the frontend. Move the CodeLoader now.
2024-03-29 02:24:53 -07:00
Ryan Houdek aa26b6288e Merge pull request #3522 from alyssarosenzweig/ra/cmpxchg8
OpcodeDispatcher: eliminate branch in cmpxchg pair
2024-03-27 21:56:19 -07:00
Mai 3d31291c3d Merge pull request #3510 from Sonicadvance1/fix_pthread_memleak
Linux/Threads: Fixes a stack memory leak for pthreads
2024-03-27 21:38:44 -04:00
Ryan Houdek 624bc3fce5 Merge pull request #3520 from Sonicadvance1/sleep_process
FEXLoader: Add a way to sleep a process on startup
2024-03-27 18:35:06 -07:00
Alyssa Rosenzweig d1722ab119 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-27 12:40:27 -04:00
Alyssa Rosenzweig 61758ea47d OpcodeDispatcher: eliminate branch in cmpxchg pair
In the old case:

* if we take the branch, 1 instruction
* if we don't take the branch, 3 instruction
* branch predictor fun
* 3 instructions of icache pressure

In the new case:

* unconditionally 2 instructions
* no branch predictor dependence
* 2 instructions of icache pressure

This should not be non-neglibly worse, and it simplifies things for RA.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-27 12:40:06 -04:00
Ryan Houdek 7b74ca1931 Merge pull request #3514 from alyssarosenzweig/opt/demon
rewrite Demon Addition Adjust (DAA) and other demonic opcodes
2024-03-26 23:24:00 -07:00
Ryan Houdek 24fd28ed9e Merge pull request #3511 from Sonicadvance1/more_tso_levers
FEXCore: Adds more TSO control levers
2024-03-26 23:23:41 -07:00
Ryan Houdek 970d5d5b13 Merge pull request #3509 from Sonicadvance1/allow_telemetry_redirect
Telemetry: Allow redirecting directory that data is written to
2024-03-26 23:23:05 -07:00
Ryan Houdek 79454ed8a6 Merge pull request #3507 from Sonicadvance1/fd_tracking_check
FEXLoader: Add some debug-only tracking for FEX owned FDs
2024-03-26 23:22:29 -07:00
Ryan Houdek 7f90ca53f7 Merge pull request #3505 from Sonicadvance1/telemetry_noncanonical
Telemetry: Adds tracker for non-canonical memory access crash
2024-03-26 23:21:32 -07:00
Mai 542f454630 Merge pull request #3517 from Sonicadvance1/remove_mman
FEXCore: Removes vestigial mman SMC checking
2024-03-26 23:28:05 -04:00
Ryan Houdek 4ea6305940 Merge pull request #3521 from neobrain/fix_libfwd_vkcreateinstance
Library Forwarding/vulkan: Fix query of vkCreateInstance function pointer
2024-03-26 08:59:23 -07:00
Tony Wasserka 4d8ffa2abb Library Forwarding/vulkan: Fix query of vkCreateInstance function pointer
The Vulkan specification states that querying "global commands" like
vkCreateInstance with a non-NULL instance is undefined behavior. Indeed, some
implementations will return null pointers in such cases.

Instead, we can drop the query from DoSetupWithInstance altogether, since
the library initializer will load the function pointer using dlsym instead.

Fixes #3519.
2024-03-26 16:25:51 +01:00
Ryan Houdek ade0c46845 FEXLoader: Add a way to sleep a process on startup
I find myself reimplementing this nearly monthly. Actually codify it so
I can stop reimplementing it.
2024-03-26 07:48:09 -07:00
Ryan Houdek 1450c92b60 Merge pull request #3518 from pmatos/20MFix
Put <20M in double quotes to avoid truncate error
2024-03-26 05:10:45 -07:00
Paulo Matos 53f02ee869 Put <20M in double quotes to avoid truncate error
Avoids bash assuming < is redirection.
See error in https://github.com/FEX-Emu/FEX/actions/runs/8434045431/job/23096421079#step:18:16
2024-03-26 11:47:57 +00:00
Ryan Houdek 6f29e75f67 FEXCore: Removes vestigial mman SMC checking
This wasn't actually wired up to anything ever since some refactoring
occured two years ago.
2024-03-26 02:56:26 -07:00
Ryan Houdek c1c797bcba FEXConfig: Add new TSO levers
Nice and convenient when testing applications.
2024-03-26 02:50:54 -07:00
Alyssa Rosenzweig bbc232741b InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 19:43:20 -04:00
Alyssa Rosenzweig dfe0bdd7f2 OpcodeDispatcher: rewrite DAS
exhaustively checked against the Intel pseudocode since this is tricky:

  def intel(AL, CF, AF):
      old_AL = AL
      old_CF = CF
      CF = False

      if (AL & 0x0F) > 9 or AF:
          Borrow = AL < 6
          AL = (AL - 6) & 0xff
          CF = old_CF or Borrow
          AF = True
      else:
          AF = False

      if (old_AL > 0x99) or old_CF:
          AL = (AL - 0x60) & 0xff
          CF = True

      return (AL & 0xff, CF, AF)

  def fex(AL, CF, AF):
      AF = AF | ((AL & 0xf) > 9)
      CF = CF | (AL > 0x99)
      NewCF = CF | (AF if (AL < 6) else CF)
      AL = (AL - 6) if AF else AL
      AL = (AL - 0x60) if CF else AL
      return (AL & 0xff, NewCF, AF)

  for AL in range(256):
      for CF in [False, True]:
          for AF in [False, True]:
              ref = intel(AL, CF, AF)
              test = fex(AL, CF, AF)
              print(AL, "CF" if CF else "", "AF" if AF else "", ref, test)
              assert(ref == test)

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 19:43:10 -04:00
Alyssa Rosenzweig e26481e3cc OpcodeDispatcher: simplify AAM
in the area.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 19:43:10 -04:00
Alyssa Rosenzweig 86b5a2f352 OpcodeDispatcher: simplify AAD
noticed in the area.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 19:43:10 -04:00
Alyssa Rosenzweig 2bf880c43a OpcodeDispatcher: rewrite AAS
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 19:43:10 -04:00
Alyssa Rosenzweig 583d4f8f94 OpcodeDispatcher: factor out CalculateAFForDecimal
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 19:43:10 -04:00
Alyssa Rosenzweig 3ca2c4377f OpcodeDispatcher: rewrite AAA
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 19:43:10 -04:00
Ryan Houdek 32ec4a3c81 Merge pull request #3508 from Sonicadvance1/stop_using_vla
ELFParser: Stop using a VLA
2024-03-25 12:18:37 -07:00
Ryan Houdek 76983476b9 Merge pull request #3504 from Sonicadvance1/fix_loop_a16
OpcodeDispatcher: Fixes 32-bit mode LOOP RCX register usage
2024-03-25 12:18:14 -07:00
Alyssa Rosenzweig 150af80f3f Merge pull request #3512 from Sonicadvance1/panic_spilling_block
InstcountCI: Adds a block that is causing panic spilling
2024-03-25 13:13:48 -04:00
Alyssa Rosenzweig a8b59c16d6 Merge pull request #3513 from Sonicadvance1/panic_spilling_rip
RA: Adds RIP when a block panic spills
2024-03-25 13:13:15 -04:00
Alyssa Rosenzweig 949717a95f OpcodeDispatcher: rewrite DAA implementation
Based on https://www.righto.com/2023/01/

New implementation is branchless, which is theoretically easier to RA. It's also
massively simpler which is good for a demon opcode.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 13:00:59 -04:00
Alyssa Rosenzweig 693d86dd67 OpcodeDispatcher: add SetAFAndFixup helper
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 12:59:19 -04:00
Alyssa Rosenzweig ea4fce7a43 InstcountCI: add flagm primary 32-bit
track the demon opcodes

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-25 12:49:00 -04:00
Ryan Houdek 3034edb0aa RA: Adds RIP when a block panic spills
I find myself adding this every time I find a game that panic spills.
Let's just print it out.
2024-03-24 17:11:29 -07:00
Ryan Houdek c025039651 InstcountCI: Adds a block that is causing panic spilling 2024-03-24 17:10:00 -07:00
Ryan Houdek 64f47d1ec2 FEXCore: Adds more TSO control levers
Lets use control vector loadstores and memcpy/memset TSO visibility.
This just gives us a bit more configuration rather than TSO off or on.
2024-03-24 16:34:18 -07:00
Ryan Houdek ea31363221 Linux/Threads: Fixes a stack memory leak for pthreads
Same situation as the last stack leak memory fix, this is fairly tricky
since it is dealing with stack pivoting. Fixes the memory leak around
pthread stack allocations, making memory usage lower for applications
that constantly spin-up and destroy threads (Like Steam).

We need to let glibc allocate a minimum sized stack (128KB and we can't
control it) to work around a race condition with DTV/TLS regions. This
means we need to do a stack pivot once the thread starts executing.

We also need to be careful because the `PThread` object is deleted
inside of the execution thread, which was resulting in a use-after-free
bug.

There are definitely some more memory leaks that I'm still fighting, and I have
noticed in my abusive thread creation program that we might want to
change some jemalloc options to more aggressively cut down on residency.
This is just one out of many.
2024-03-24 05:22:22 -07:00
Ryan Houdek 70befc216f Telemetry: Allow redirecting directory that data is written to
This will be necessary
2024-03-24 00:47:35 -07:00
Ryan Houdek 50f62663ac ELFParser: Stop using a VLA
Clang-18 complains about this, use a vector instead.
2024-03-22 22:51:57 -07:00
Ryan Houdek 60755acef0 FEXLoader: Add some debug-only tracking for FEX owned FDs
I remember seeing some application last year where they closed a FEX
owned FD but now I don't remember what it was. This can really mess us
up so add some debug tracking so we can try and find it again.

Might be something specifically around flatpack, appimage, or chrome's
sandbox. I have some ideas about how to work around these problems if
they crop up but need to find the problem applications again.
2024-03-22 22:49:26 -07:00
Mai 002ca360f8 Merge pull request #3506 from Sonicadvance1/telemetry_rename
Telemetry: Rename old file instead of copying
2024-03-23 00:24:02 -04:00
Ryan Houdek 4952b2e16c Telemetry: Rename old file instead of copying
Since we do an immediate overwrite of the file we are copying, we can
instead do a rename. Failure on rename is fine, will either mean the
telemetry file didn't exist initially, or some other permission error so
the telemetry will get lost regardless.
2024-03-21 22:51:20 -07:00
Ryan Houdek cccf263080 InstCountCI: Update for Telemetry offset changes 2024-03-21 21:10:03 -07:00
Ryan Houdek 5a35e119fe Telemetry: Adds tracker for non-canonical memory access crash
This may be useful for tracking TSO faulting when it manages to fetch
stale data. While most TSO crashes are due to nullptr dereferences, this
can still check for the corruption case.
2024-03-21 20:47:36 -07:00
Ryan Houdek 9ab930cb26 unittests/ASM: Adds tests for loop instruction address size overrides
32-bit test would fail if the 16-bit address size override wasn't
respected.
2024-03-21 20:18:43 -07:00
Ryan Houdek 824f122680 OpcodeDispatcher: Fixes 32-bit mode LOOP RCX register usage
In 64-bit mode, the LOOP instruction's RCX register usage is 64-bit or
32-bit.
In 32-bit mode, the LOOP instruction's RCX register usage is 32-bit or
16-bit.

FEX wasn't handling the 16-bit case at all which was causing the LOOP
instruction to effectively always operate at 32-bit size. Now this is
correctly supported, and it also stops treating the operation as 64-bit.
2024-03-21 20:13:15 -07:00
Ryan Houdek 8852d94416 Merge pull request #3503 from alyssarosenzweig/opt/loop
OpcodeDispatcher: optimize LOOP/N/E
2024-03-21 20:05:50 -07:00
Mai 0c24aea27e Merge pull request #3502 from Sonicadvance1/remove_termux
Removes false termux support
2024-03-21 12:45:01 -04:00
Alyssa Rosenzweig 82ba16c6ed OpcodeDispatcher: optimize LOOP/N/E
Don't clobber NZCV.

Before/after assembly from the Primary_E1 unit test:

< 4340: [INFO] cset w20, ne
< 4340: [INFO] mrs x21, nzcv
< 4340: [INFO] cmp x5, #0x0 (0)
< 4340: [INFO] cset x22, ne
< 4340: [INFO] and x20, x22, x20
< 4340: [INFO] msr nzcv, x21
< 4340: [INFO] cbnz x20, #+0x8 (addr 0xffff896f8084)
< 4340: [INFO] b #+0x1c (addr 0xffff896f809c)
< 4340: [INFO] ldr x0, pc+8 (addr 0xffff896f808c)
---
> 4340: [INFO] csel x20, x5, xzr, ne
> 4340: [INFO] cbnz x20, #+0x8 (addr 0xfffed7308070)
> 4340: [INFO] b #+0x1c (addr 0xfffed7308088)
> 4340: [INFO] ldr x0, pc+8 (addr 0xfffed7308078)

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-21 12:08:40 -04:00
Ryan Houdek 45ea0cd782 Removes false termux support
This was a funny joke that this was here, but it is fundamentally
incompatible with what we're doing. All those users are running proot
anyway because of how broken running under termux directly is.

Just remove this from here.
2024-03-20 22:04:32 -07:00
Ryan Houdek 6ce366ef35 Merge pull request #3499 from Sonicadvance1/overlapping_memcpy_unittest
unittests/ASM: Adds a test for overlapping memcpy using rep movs
2024-03-20 19:52:19 -07:00
Ryan Houdek 862d63adf2 unittests/ASM: Adds a test for overlapping memcpy using rep movs
Caused by #3478

This was missed in the review that it could cause issues. bylaws already
has a fix incoming that will get this unit test working.
2024-03-20 18:44:36 -07:00
Ryan Houdek 167896dc9d Merge pull request #3501 from bylaws/memcpy
FEXCore: Fallback to the memcpy slow path for overlaps within 32 bytes
2024-03-20 18:42:56 -07:00
Billy Laws 12fb26f9c0 Update InstCountCI 2024-03-20 21:08:59 +00:00
Billy Laws d490cb1b79 FEXCore: Fallback to the memcpy slow path for overlaps within 32 bytes
Take e.g a forward rep movsb copy from addr 0 to 1, the expected
behaviour since this is a bytewise copy is:
before: aaabbbb...
after: aaaaaaa...
but by copying in 32-byte chunks we end up with:
after: aaaabbbb...
due to the self overwrites not occuring within a single 32 bit copy.
2024-03-20 20:54:19 +00:00
Billy Laws 94fecb9dad FEXCore: Remove needless alignment checks for the mem{cpy,set} fastpath 2024-03-20 20:54:09 +00:00
Ryan Houdek 7dcacfe990 Merge pull request #3478 from bylaws/memcpy
FEXCore: Add non-atomic Memcpy and Memset IR fast paths
2024-03-18 18:56:44 -07:00
Billy Laws 29b05f6b90 Update InstCountCI 2024-03-18 23:30:19 +00:00
Billy Laws 8d4d8fe3e5 FEXCore: Add non-atomic Memcpy and Memset IR fast paths
When TSO is disabled, vector LDP/STP can be used for a two
instruction 32 byte memory copy which is significantly faster than the
current byte-by-byte copy. Performing two such copies directly after
oneanother also marginally increases copy speed for all sizes >=64.
2024-03-18 23:28:50 +00:00
Ryan Houdek ab8ee64352 Merge pull request #3497 from Sonicadvance1/movmaskb_constant
JIT: Optimize pmovmaskb with a named vector constant
2024-03-18 16:08:40 -07:00
Alyssa Rosenzweig 2a9fcc6a66 Merge pull request #3492 from Sonicadvance1/implement_prefetch
OpcodeDispatcher: Implement support for the various prefetch instructions
2024-03-18 07:49:47 -04:00
Ryan Houdek 20da1e4244 InstcountCI: Update for pmovmaskb 2024-03-17 18:52:21 -07:00
Ryan Houdek fd391b1b18 JIT: Optimize pmovmaskb with a named vector constant
I was looking at some other JIT overheads and this cropped up as some
overhead. Instead of materializing a constant using mov+movk+movk+movk,
load it from the named vector constant array.

In a micro-benchmark this improved performance by 34%.
In bytemark this improved on subbench by 0.82%
2024-03-17 18:40:46 -07:00
Mai ba3029b1f6 Merge pull request #3495 from Sonicadvance1/implement_rdpid
OpcodeDispatcher: Implement rdpid
2024-03-15 17:29:12 -04:00
Ryan Houdek 6757a80365 InstcountCI: Update for prefetch changes 2024-03-15 13:20:28 -07:00
Ryan Houdek f79991a9d8 OpcodeDispatcher: Implement rdpid
Missed this instruction when implementing rdtscp. Returns the same ID
result in a register just like rdtscp, but without the cycle counter
results. Doesn't touch any flags just like rdtscp.
2024-03-14 20:07:58 -07:00
Ryan Houdek ca6b2e43e6 Merge pull request #3491 from alyssarosenzweig/rclse/waw
RCLSE: Optimize store-after-store
2024-03-14 03:23:05 -07:00
Ryan Houdek 8a3d08e1d8 Merge pull request #3483 from neobrain/refactor_stealmemoryregion
Allocator: Cleanup StealMemoryRegions implementation
2024-03-14 03:21:09 -07:00
Ryan Houdek cd2a6ce820 Merge pull request #3469 from alyssarosenzweig/opt/df
Optimize DF representation
2024-03-14 03:18:41 -07:00
Ryan Houdek 4e269d8b80 Merge pull request #3494 from neobrain/fix_libfwd_float_as_int
Library Forwarding: Don't map float/double to fixed-size integers
2024-03-14 03:11:11 -07:00
Tony Wasserka 552e76c001 Library Forwarding: Don't map float/double to fixed-size integers
Fixes #3455.
2024-03-14 10:14:57 +01:00
Mai caff3cb799 Merge pull request #3493 from Sonicadvance1/bug_for_3478
unittests/ASM: Implements a unit test for #3478
2024-03-13 22:57:39 -04:00
Ryan Houdek 0d33dacc37 unittests/ASM: Implements a unit test for #3478
This unit test recreates the error condition that #3478 causes.
With a string operation that is a backwards copy then the optimization
will read past the end of the page and result in a crash.

Seemingly only happens with backwards string operations, but test
forward and backward in this test.
2024-03-13 18:36:19 -07:00
Ryan Houdek ba7b69eea2 InstCountCI: Adds prefetch addressing limits 2024-03-12 21:38:28 -07:00
Ryan Houdek 8056bee82b OpcodeDispatcher: Implement support for the various prefetch instructions
x86 has a few prefetch instructions.
- prefetch - One of two classic 3DNow! instructions
   - Prefetch in to L1 data cache
- prefetchw - One of two classic 3DNow! instructions
   - Implies prefetch in to L1 data cache
   - Prefetch cacheline with intent to write and exclusive ownership

- prefetchnta
   - Prefetch non-temporal data in respect to /all/ cache levels
   - Assumes inclusive caches?
- prefetch{t0,t1,t2}
   - Prefetch data with respect to each cache level
   - T0 = L1 and higher
   - T1 = L2 and higher
   - T2 = L3 and higher

**Some silly duplicates**
- prefetchwt1
   - Duplicate of prefetchw but explicitly L1 data cache
- prefetch_exclusive
   - Duplicate of prefetch

God Of War 2018 uses prefetchw as a hint for exclusive ownership of the
cacheline in some very aggressive spin-loops. Let's implement the
operations to help it along.
2024-03-12 21:37:31 -07:00
Ryan Houdek cc635a54f8 IR: Implements support for prefetch operation 2024-03-12 21:19:50 -07:00
Ryan Houdek 217d9d8c50 ARMEmitter: Fixes prfm with negative or unaligned offsets 2024-03-12 21:18:23 -07:00
Tony Wasserka a047ac1699 Allocator: Test CollectMemoryGaps instead of StealMemoryRegions and restore the original interfaces 2024-03-12 10:49:31 +01:00
Tony Wasserka bb0b114fc8 Allocator: Miscellaneous cleanups 2024-03-12 10:49:30 +01:00
Tony Wasserka ccd6c15316 Allocator: Use std::from_chars instead of parsing digits manually 2024-03-12 10:49:30 +01:00
Tony Wasserka 0a1fe1c8c2 Allocator: Parse process mappings per-line instead of per-character 2024-03-12 10:49:30 +01:00
Tony Wasserka f43fe5fd63 Allocator: Stop parsing more eagerly
This is a soft-revert of eaf83aa. That change is no longer needed, since the
stack special case is handled externally now.
2024-03-12 10:49:30 +01:00
Tony Wasserka dce9f651fd Allocator: Split off memory gap collection to a separate function
This function can be unit-tested more easily, and the stack special is more
cleanly handled as a post-collection step.

There is a minor functional change: The stack special case didn't trigger
previously if the range end was within the stack mapping. This is now fixed.
2024-03-12 10:49:30 +01:00
Tony Wasserka 0d71f169d0 Allocator: Adopt a more testable interface for StealMemoryRegions 2024-03-12 10:49:30 +01:00
Tony Wasserka 430ac0f70a Allocator: Fix format strings 2024-03-12 10:49:30 +01:00
Alyssa Rosenzweig 063b81da1d InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:50:31 -04:00
Alyssa Rosenzweig 7629007cfa OpcodeDispatcher: allow upper garbage on STOS
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:50:31 -04:00
Alyssa Rosenzweig 03c6abdad4 OpcodeDispatcher: optimize DF add
fuse the shift the right way

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:50:31 -04:00
Alyssa Rosenzweig c99cbe6d0a JIT: switch DF representation
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:50:31 -04:00
Alyssa Rosenzweig e3ee65e491 OpcodeDispatcher: use transformed DF for memset/memcpy
Use the 1/-1 representation instead of 0/1. This will be better by the end of
the series.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:50:31 -04:00
Alyssa Rosenzweig aee00f524c OpcodeDispatcher: use DF retrieval helpers
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:50:31 -04:00
Alyssa Rosenzweig a76321c6c1 OpcodeDispatcher: add DF retrieval helpers
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:50:31 -04:00
Alyssa Rosenzweig f7586f4459 CoreState: use x86 enums for readability
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:50:31 -04:00
Ryan Houdek e33a76a2ad Merge pull request #3489 from Sonicadvance1/linux_v6.8
Linux: Expose support for v6.8
2024-03-11 15:48:37 -07:00
Ryan Houdek c37a12e806 Merge pull request #3490 from Sonicadvance1/disable_assert
Disable assert in release
2024-03-11 15:48:18 -07:00
Alyssa Rosenzweig ed59f73a65 InstCountCI: Update
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:41:33 -04:00
Alyssa Rosenzweig 92f31648b9 RCLSE: optimize out pointless stores
can help a lot of x86 code because x86 is 2-address and a64 is 3-address, so x86
ends up with piles of movs that end up dead after translation

It's not a win across the board because our RA isn't aware of tied registers so
sometimes we regress moves. But it's a win on average, and the RA bits can be
improved with time.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:41:23 -04:00
Alyssa Rosenzweig 85f8ad3842 JIT: fix sha256msg1 encoding
botched move in the !tied reg case.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:41:23 -04:00
Alyssa Rosenzweig 64f71d87bb unittests: disable rdtsc test on sim
Patch written by Sonicadvance1. Unclear how this wasn't already broken, but we
need this to keep CI happy with the rest of this series.

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2024-03-11 18:41:23 -04:00
Ryan Houdek ff0c7637c9 Merge pull request #3421 from pmatos/AddressingModes32
Improve 32bit ld/st addressing mode propagation
2024-03-11 15:20:20 -07:00
Ryan Houdek 54403e2146 Disable assert in release
Arguments and conditional doesn't get optimized out in release builds
for the inline function call versus the define.

Was showing up an annoying amount of time when testing.
2024-03-10 22:01:50 -07:00
Ryan Houdek d8202335e0 Linux: Expose support for v6.8
The new syscalls for futexes are the most interesting part
2024-03-10 15:48:55 -07:00
Ryan Houdek 9ec20c4bef Linux/Ioctls: Update ioctl emulation for v6.8
- v3d added an ioctl
- drm base added a new ioctl
- pvr and xe are new drivers in v6.8
2024-03-10 15:46:21 -07:00
Ryan Houdek ce7acd9b71 External/drm-headers: Update to v6.8 2024-03-10 15:45:34 -07:00
Ryan Houdek 8a607135fd Linux: Update syscalls for v6.8 2024-03-10 15:22:51 -07:00
Tony Wasserka 26a66790ab Merge pull request #3486 from neobrain/fix_libfwd_accidental_copy
Library Forwarding: Fix accidental data copying when converting from host to guest layout
2024-03-08 12:40:44 +01:00
Mai 6d94d79409 Merge pull request #3485 from Sonicadvance1/nouveau_ioctl
IoctlEmulation: Add missing nouveau ioctl
2024-03-07 07:57:47 -05:00
Tony Wasserka 2359a9899c Library Forwarding: Fix accidental data copying when converting from host to guest layout 2024-03-07 10:52:13 +01:00
Ryan Houdek aeb41e9ae2 IoctlEmulation: Add missing nouveau ioctl
The NVIF ioctl isn't publicly described in the nouveau headers and it is
required for anything to work with Nouveau.

Pass the ioctl command through without modification and hope that this
ioctl is architecture agnostic.
2024-03-05 16:05:13 -08:00
Tony Wasserka b892da72f3 Merge pull request #3484 from neobrain/feature_catch2_v3
Externals: Update Catch2 to v3.5.3
2024-03-05 19:30:30 +01:00
Paulo Matos a86f2d3e2c Improve 32bit constant usage in memory addressing
Folds reg+const memory address into addressing mode,
if the constant is within 16Kb.
Update instcountci files.
Add test 32Bit_ASM/FEX_bugs/SubAddrBug.asm
2024-03-05 14:01:32 +00:00
Tony Wasserka 6edba49784 Update Catch2 to v3.5.3 2024-03-05 12:15:29 +01:00
208 changed files with 9659 additions and 7406 deletions

No files matched your search

+1 -1
View File
@@ -237,7 +237,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
+1 -1
View File
@@ -171,7 +171,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
+1 -1
View File
@@ -90,7 +90,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+1 -1
View File
@@ -121,7 +121,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+1 -1
View File
@@ -106,7 +106,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
-13
View File
@@ -26,7 +26,6 @@ option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
@@ -163,18 +162,6 @@ if (NOT ENABLE_OFFLINE_TELEMETRY)
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
endif()
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
add_definitions(-DTERMUX_BUILD=1)
set(TERMUX_BUILD 1)
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
set(ENABLE_JEMALLOC FALSE)
# Termux builds can't rely on X11 packages
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
set(BUILD_FEXCONFIG FALSE)
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
+1 -1
+19 -1
View File
@@ -48,6 +48,7 @@ namespace DefaultValues {
PATH_CONFIG_DIR_GLOBAL,
PATH_CONFIG_FILE_LOCAL,
PATH_CONFIG_FILE_GLOBAL,
PATH_CONFIG_TELEMETRY_FOLDER,
PATH_LAST,
};
static std::array<fextl::string, Paths::PATH_LAST> Paths;
@@ -64,6 +65,22 @@ namespace DefaultValues {
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
}
fextl::string const& GetTelemetryDirectory() {
auto &Path = Paths[PATH_CONFIG_TELEMETRY_FOLDER];
if (Path.empty()) {
FEX_CONFIG_OPT(TelemetryDirectory, TELEMETRYDIRECTORY);
if (!TelemetryDirectory().empty()) {
Path = TelemetryDirectory;
Path += "/";
}
else {
Path = Config::GetDataDirectory() + "Telemetry/";
}
}
return Path;
}
fextl::string const& GetDataDirectory() {
return Paths[PATH_DATA_DIR];
}
@@ -113,7 +130,7 @@ namespace DefaultValues {
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
static FEXCore::Config::Layer *Meta{};
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
constexpr std::array<FEXCore::Config::LayerType, 10> LoadOrder = {
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
FEXCore::Config::LayerType::LAYER_MAIN,
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP,
@@ -121,6 +138,7 @@ namespace DefaultValues {
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
FEXCore::Config::LayerType::LAYER_ARGUMENTS,
FEXCore::Config::LayerType::LAYER_USER_OVERRIDE,
FEXCore::Config::LayerType::LAYER_ENVIRONMENT,
FEXCore::Config::LayerType::LAYER_TOP
};
+33 -3
View File
@@ -368,6 +368,14 @@
"File to write FEX output to.",
"[stdout, stderr, server, <Filename>]"
]
},
"TelemetryDirectory": {
"Type": "str",
"Default": "",
"Desc": [
"Redirects the telemetry folder that FEX usually writes to.",
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
]
}
},
"Hacks": {
@@ -379,9 +387,8 @@
"Desc": [
"Checks code for modification before execution.",
"\tnone: No checks",
"\tmtrack: Page tracking based invalidation",
"\tfull: Validate code before every run (slow)",
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
"\tmtrack: Page tracking based invalidation (default)",
"\tfull: Validate code before every run (slow)"
]
},
"TSOEnabled": {
@@ -392,6 +399,21 @@
"Highly likely to break any multithreaded application if disabled."
]
},
"VectorTSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
]
},
"MemcpySetTSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
"Only affects REP MOVS and REP STOS instructions"
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
@@ -439,6 +461,14 @@
"Hides the hypervisor CPUID bit when set.",
"Should only be used for applications that have issues with this set."
]
},
"StartupSleep": {
"Type": "uint32",
"Default": "0",
"Desc": [
"Sleeps the process at startup for a duration of seconds.",
"Useful if an application crashes too quickly to attach a debugger."
]
}
},
"Misc": {
@@ -66,4 +66,8 @@ namespace FEXCore::Context {
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
return CPUID.RunFunctionName(Function, Leaf, CPU);
}
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState *Thread, uintptr_t Address) const {
return Thread->CPUBackend->IsAddressInCodeBuffer(Address);
}
}
+5 -1
View File
@@ -13,6 +13,7 @@
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
@@ -183,6 +184,9 @@ namespace FEXCore::Context {
void MarkMemoryShared(FEXCore::Core::InternalThreadState *Thread) override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState *Thread, uintptr_t Address) const override;
// returns false if a handler was already registered
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr);
@@ -277,7 +281,7 @@ namespace FEXCore::Context {
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
LOGMAN_THROW_A_FMT(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ThreadRemoveCodeEntry(Thread, GuestRIP);
@@ -3762,7 +3762,12 @@ public:
}
else {
if (MemSrc.MetaType.ImmType.Index == ARMEmitter::IndexType::OFFSET) {
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
if ((MemSrc.MetaType.ImmType.Imm & 0b111) || MemSrc.MetaType.ImmType.Imm < 0) {
prfum<IndexType::OFFSET>(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
}
else {
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
}
}
else {
LOGMAN_MSG_A_FMT("Unexpected loadstore index type");
+3 -1
View File
@@ -2,8 +2,8 @@
#include "FEXCore/IR/IR.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include <FEXCore/Core/CPUBackend.h>
#ifndef _WIN32
#include <sys/prctl.h>
@@ -27,6 +27,8 @@ constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant:
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB_UPPER
};
constexpr static auto PSHUFLW_LUT {
+1 -1
View File
@@ -694,7 +694,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 19) | // MPX MAWAU
(0 << 20) | // MPX MAWAU
(0 << 21) | // MPX MAWAU
(0 << 22) | // RDPID Read Processor ID
(1 << 22) | // RDPID Read Processor ID
(0 << 23) | // Reserved
(0 << 24) | // Reserved
(0 << 25) | // CLDEMOTE
+12 -10
View File
@@ -12,6 +12,7 @@ $end_info$
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
@@ -25,23 +26,19 @@ $end_info$
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Utils/Allocator.h"
#include "Utils/Allocator/HostAllocator.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CodeLoader.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/HLE/SourcecodeResolver.h>
#include <FEXCore/HLE/Linux/ThreadManagement.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/File.h>
@@ -167,6 +164,7 @@ namespace FEXCore::Context {
case X86State::RFLAG_ZF_RAW_LOC:
case X86State::RFLAG_SF_RAW_LOC:
case X86State::RFLAG_OF_RAW_LOC:
case X86State::RFLAG_DF_RAW_LOC:
// Intentionally do nothing.
// These contain multiple bits which can corrupt other members when compacted.
break;
@@ -215,6 +213,11 @@ namespace FEXCore::Context {
uint32_t AF = ((Frame->State.af_raw ^ PFByte) & (1 << 4)) ? 1 : 0;
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
// DF is pretransformed, undo the transform from 1/-1 back to 0/1
uint8_t DFByte = Frame->State.flags[X86State::RFLAG_DF_RAW_LOC];
if (DFByte & 0x80)
EFLAGS |= 1 << X86State::RFLAG_DF_RAW_LOC;
return EFLAGS;
}
@@ -238,6 +241,10 @@ namespace FEXCore::Context {
// PF is inverted in our internal representation.
Frame->State.pf_raw = (EFLAGS & (1U << i)) ? 0 : 1;
break;
case X86State::RFLAG_DF_RAW_LOC:
// DF is encoded as 1/-1
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 0xff : 1;
break;
default:
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 1 : 0;
break;
@@ -468,7 +475,6 @@ namespace FEXCore::Context {
Thread->LookupCache->ClearCache();
Thread->CPUBackend->ClearCache();
Thread->DebugStore.clear();
}
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, IR::IREmitter *IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
@@ -947,9 +953,6 @@ namespace FEXCore::Context {
// Only the lookup cache is cleared here, so that old code can keep running until next compilation
std::lock_guard<std::recursive_mutex> lkLookupCache(Thread->LookupCache->WriteLock);
Thread->LookupCache->ClearCache();
// DebugStore also needs to be cleared
Thread->DebugStore.clear();
}
}
}
@@ -965,7 +968,6 @@ namespace FEXCore::Context {
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
Thread->DebugStore.erase(GuestRIP);
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
}
@@ -2,8 +2,8 @@
#pragma once
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/fextl/memory.h>
#ifdef VIXL_SIMULATOR
+6 -6
View File
@@ -20,8 +20,8 @@ $end_info$
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/set.h>
#include <FEXHeaderUtils/TypeDefines.h>
namespace FEXCore::Frontend {
#include "Interface/Core/VSyscall/VSyscall.inc"
@@ -1126,11 +1126,11 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
// Entry is a jump target
BlocksToDecode.emplace(PC);
uint64_t CurrentCodePage = PC & FHU::FEX_PAGE_MASK;
uint64_t CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
fextl::set<uint64_t> CodePages = { CurrentCodePage };
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
AddContainedCodePage(PC, CurrentCodePage, FEXCore::Utils::FEX_PAGE_SIZE);
if (MaxInst == 0) {
MaxInst = CTX->Config.MaxInstPerBlock;
@@ -1156,8 +1156,8 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
auto OpMinAddress = RIPToDecode + PCOffset;
auto OpMaxAddress = OpMinAddress + MAX_INST_SIZE;
auto OpMinPage = OpMinAddress & FHU::FEX_PAGE_MASK;
auto OpMaxPage = OpMaxAddress & FHU::FEX_PAGE_MASK;
auto OpMinPage = OpMinAddress & FEXCore::Utils::FEX_PAGE_MASK;
auto OpMaxPage = OpMaxAddress & FEXCore::Utils::FEX_PAGE_MASK;
if (OpMinPage != CurrentCodePage) {
CurrentCodePage = OpMinPage;
@@ -1230,7 +1230,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
}
for (auto CodePage : CodePages) {
AddContainedCodePage(PC, CodePage, FHU::FEX_PAGE_SIZE);
AddContainedCodePage(PC, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
// sort for better branching
@@ -2,9 +2,8 @@
#include "Common/SoftFloat.h"
#include "Common/SoftFloat-3e/softfloat.h"
#include <FEXCore/IR/IR.h>
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
#include "Interface/IR/IR.h"
namespace FEXCore::CPU {
FEXCORE_PRESERVE_ALL_ATTR
@@ -7,7 +7,6 @@
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::IR {
class IRListView;
@@ -5,6 +5,7 @@ tags: backend|arm64
$end_info$
*/
#include "FEXCore/IR/IR.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
@@ -299,6 +300,31 @@ DEF_OP(SubNZCV) {
}
}
DEF_OP(CmpPairZ) {
auto Op = IROp->C<IR::IROp_CmpPairZ>();
const uint8_t OpSize = IROp->Size;
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
// Save NZCV
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
// Compare, setting Z and clobbering NzCV
const auto Src1 = GetRegPair(Op->Src1.ID());
const auto Src2 = GetRegPair(Op->Src2.ID());
cmp(EmitSize, Src1.first, Src2.first);
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
// Restore NzCV
if (CTX->HostFeatures.SupportsFlagM) {
rmif(TMP1, 0, 0xb /* NzCV */);
} else {
cset(ARMEmitter::Size::i32Bit, TMP2, ARMEmitter::Condition::CC_EQ);
bfi(ARMEmitter::Size::i32Bit, TMP1, TMP2, 30 /* lsb: Z */, 1);
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
}
}
DEF_OP(CarryInvert) {
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
cfinv();
@@ -308,7 +334,7 @@ DEF_OP(RmifNZCV) {
auto Op = IROp->C<IR::IROp_RmifNZCV>();
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
rmif(GetReg(Op->Src.ID()).X(), Op->Rotate, Op->Mask);
rmif(GetZeroableReg(Op->Src).X(), Op->Rotate, Op->Mask);
}
DEF_OP(SetSmallNZV) {
@@ -376,6 +402,24 @@ DEF_OP(CondAddNZCV) {
}
}
DEF_OP(CondSubNZCV) {
auto Op = IROp->C<IR::IROp_CondSubNZCV>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
ARMEmitter::StatusFlags Flags = (ARMEmitter::StatusFlags)Op->FalseNZCV;
uint64_t Const = 0;
auto Src1 = GetZeroableReg(Op->Src1);
if (IsInlineConstant(Op->Src2, &Const)) {
ccmp(EmitSize, Src1, Const, Flags, MapSelectCC(Op->Cond));
} else {
ccmp(EmitSize, Src1, GetReg(Op->Src2.ID()), Flags, MapSelectCC(Op->Cond));
}
}
DEF_OP(Neg) {
auto Op = IROp->C<IR::IROp_Neg>();
const uint8_t OpSize = IROp->Size;
@@ -743,6 +787,16 @@ DEF_OP(XorShift) {
eor(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
DEF_OP(XornShift) {
auto Op = IROp->C<IR::IROp_XornShift>();
const uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
eon(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
DEF_OP(Lshl) {
auto Op = IROp->C<IR::IROp_Lshl>();
const uint8_t OpSize = IROp->Size;
@@ -31,11 +31,19 @@ DEF_OP(CASPair) {
mov(EmitSize, Dst.second, TMP4.R());
}
else {
// Save NZCV so we don't have to mark this op as clobbering NZCV (the
// SupportsAtomics does not clobber atomics and this !SupportsAtomics path
// is so slow it's not worth the complexity of splitting the IR op.). We
// clobber NZCV inside the hot loop and we can't replace cmp/ccmp/b.ne with
// something NZCV-preserving without requiring an extra instruction.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
ARMEmitter::SingleUseForwardLabel LoopExpected;
Bind(&LoopTop);
// This instruction sequence must be synced with HandleCASPAL_Armv8.
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
cmp(EmitSize, TMP2, Expected.first);
ccmp(EmitSize, TMP3, Expected.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
@@ -54,6 +62,9 @@ DEF_OP(CASPair) {
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopExpected);
// Restore
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
}
}
@@ -201,7 +201,7 @@ DEF_OP(VSha256U0) {
else {
mov(VTMP1.Q(), Src1.Q());
sha256su0(VTMP1, Src2);
mov(Dst.Q(), Src1.Q());
mov(Dst.Q(), VTMP1.Q());
}
}
@@ -9,16 +9,17 @@ $end_info$
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <aarch64/assembler-aarch64.h>
#include <aarch64/disasm-aarch64.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
@@ -56,6 +57,8 @@ public:
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
const bool HostSupportsSVE128{};
const bool HostSupportsSVE256{};
@@ -888,6 +888,14 @@ DEF_OP(StoreNZCV) {
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value.ID()));
}
DEF_OP(LoadDF) {
auto Dst = GetReg(Node);
auto Flag = X86State::RFLAG_DF_RAW_LOC;
// DF needs sign extension to turn 0x1/0xFF into 1/-1
ldrsb(Dst.X(), STATE, offsetof(FEXCore::Core::CPUState, flags[Flag]));
}
DEF_OP(LoadFlag) {
auto Op = IROp->C<IR::IROp_LoadFlag>();
auto Dst = GetReg(Node);
@@ -1166,8 +1174,10 @@ DEF_OP(LoadMemTSO) {
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
break;
}
// Half-barrier.
dmb(FEXCore::ARMEmitter::BarrierScope::ISHLD);
if (VectorTSOEnabled()) {
// Half-barrier.
dmb(FEXCore::ARMEmitter::BarrierScope::ISHLD);
}
}
}
@@ -1315,7 +1325,7 @@ DEF_OP(VLoadVectorElement) {
}
// Emit a half-barrier if TSO is enabled.
if (CTX->IsAtomicTSOEnabled()) {
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
dmb(ARMEmitter::BarrierScope::ISHLD);
}
}
@@ -1335,7 +1345,7 @@ DEF_OP(VStoreVectorElement) {
ElementSize == 16, "Invalid element size");
// Emit a half-barrier if TSO is enabled.
if (CTX->IsAtomicTSOEnabled()) {
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
}
@@ -1435,7 +1445,7 @@ DEF_OP(VBroadcastFromMem) {
}
// Emit a half-barrier if TSO is enabled.
if (CTX->IsAtomicTSOEnabled()) {
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
dmb(ARMEmitter::BarrierScope::ISHLD);
}
}
@@ -1653,8 +1663,10 @@ DEF_OP(StoreMemTSO) {
}
}
else {
// Half-Barrier.
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
if (VectorTSOEnabled()) {
// Half-Barrier.
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
}
const auto Src = GetVReg(Op->Value.ID());
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
@@ -1695,6 +1707,7 @@ DEF_OP(MemSet) {
// that the value is zero, we can optimize any operation larger than 8-bit down to 8-bit to use the MOPS implementation.
const auto Op = IROp->C<IR::IROp_MemSet>();
const bool IsAtomic = Op->IsAtomic && MemcpySetTSOEnabled();
const int32_t Size = Op->Size;
const auto MemReg = GetReg(Op->Addr.ID());
const auto Value = GetReg(Op->Value.ID());
@@ -1708,7 +1721,7 @@ DEF_OP(MemSet) {
DirectionReg = GetReg(Op->Direction.ID());
}
// If Direction == 0 then:
// If Direction > 0 then:
// MemReg is incremented (by size)
// else:
// MemReg is decremented (by size)
@@ -1729,7 +1742,7 @@ DEF_OP(MemSet) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
tbnz(DirectionReg, 1, &BackwardImpl);
}
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
@@ -1784,18 +1797,73 @@ DEF_OP(MemSet) {
}
};
const auto SubRegSize =
Size == 1 ? ARMEmitter::SubRegSize::i8Bit :
Size == 2 ? ARMEmitter::SubRegSize::i16Bit :
Size == 4 ? ARMEmitter::SubRegSize::i32Bit :
Size == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
auto EmitMemset = [&](int32_t Direction) {
const int32_t OpSize = Size;
const int32_t SizeDirection = Size * Direction;
ARMEmitter::BackwardLabel AgainInternal{};
ARMEmitter::SingleUseForwardLabel DoneInternal{};
ARMEmitter::BiDirectionalLabel AgainInternal{};
ARMEmitter::ForwardLabel DoneInternal{};
// Early exit if zero count.
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AgainInternal256Exit{};
ARMEmitter::BackwardLabel AgainInternal256{};
ARMEmitter::ForwardLabel AgainInternal128Exit{};
ARMEmitter::BackwardLabel AgainInternal128{};
if (Direction == -1) {
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
}
// Keep the counter one copy ahead, so that underflow can be used to detect when to fallback
// to the copy unit size copy loop for the last chunk.
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal128Exit);
// Fill VTMP2 with the set pattern
dup(SubRegSize, VTMP2.Q(), Value);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal256Exit);
Bind(&AgainInternal256);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
tbz(TMP1, 63, &AgainInternal256);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbz(TMP1, 63, &AgainInternal128);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
}
}
Bind(&AgainInternal);
if (Op->IsAtomic) {
if (IsAtomic) {
MemStoreTSO(Value, OpSize, SizeDirection);
}
else {
@@ -1847,8 +1915,8 @@ DEF_OP(MemSet) {
};
if (DirectionIsInline) {
// If the direction constant is set then the direction is negative.
EmitMemset(DirectionConstant ? -1 : 1);
LOGMAN_THROW_AA_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
EmitMemset(DirectionConstant);
}
else {
// Emit forward direction memset then backward direction memset.
@@ -1873,6 +1941,7 @@ DEF_OP(MemCpy) {
// Assuming non-atomicity and non-faulting behaviour, this can accelerate this implementation.
const auto Op = IROp->C<IR::IROp_MemCpy>();
const bool IsAtomic = Op->IsAtomic && MemcpySetTSOEnabled();
const int32_t Size = Op->Size;
const auto MemRegDest = GetReg(Op->AddrDest.ID());
const auto MemRegSrc = GetReg(Op->AddrSrc.ID());
@@ -1886,7 +1955,7 @@ DEF_OP(MemCpy) {
}
auto Dst = GetRegPair(Node);
// If Direction == 0 then:
// If Direction > 0 then:
// MemRegDest is incremented (by size)
// MemRegSrc is incremented (by size)
// else:
@@ -1922,7 +1991,7 @@ DEF_OP(MemCpy) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
tbnz(DirectionReg, 1, &BackwardImpl);
}
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
@@ -1943,6 +2012,10 @@ DEF_OP(MemCpy) {
ldr<ARMEmitter::IndexType::POST>(TMP4, TMP3, Size);
str<ARMEmitter::IndexType::POST>(TMP4, TMP2, Size);
break;
case 32:
ldp<ARMEmitter::IndexType::POST>(VTMP1.Q(), VTMP2.Q(), TMP3, Size);
stp<ARMEmitter::IndexType::POST>(VTMP1.Q(), VTMP2.Q(), TMP2, Size);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
break;
@@ -2049,14 +2122,69 @@ DEF_OP(MemCpy) {
const int32_t OpSize = Size;
const int32_t SizeDirection = Size * Direction;
ARMEmitter::BackwardLabel AgainInternal{};
ARMEmitter::SingleUseForwardLabel DoneInternal{};
ARMEmitter::BiDirectionalLabel AgainInternal{};
ARMEmitter::ForwardLabel DoneInternal{};
// Early exit if zero count.
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AbsPos{};
ARMEmitter::ForwardLabel AgainInternal256Exit{};
ARMEmitter::ForwardLabel AgainInternal128Exit{};
ARMEmitter::BackwardLabel AgainInternal128{};
ARMEmitter::BackwardLabel AgainInternal256{};
sub(ARMEmitter::Size::i64Bit, TMP4, TMP2, TMP3);
tbz(TMP4, 63, &AbsPos);
neg(ARMEmitter::Size::i64Bit, TMP4, TMP4);
Bind(&AbsPos);
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
tbnz(TMP4, 63, &AgainInternal);
if (Direction == -1) {
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
sub(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32 - Size);
}
// Keep the counter one copy ahead, so that underflow can be used to detect when to fallback
// to the copy unit size copy loop for the last chunk.
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal128Exit);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal256Exit);
Bind(&AgainInternal256);
MemCpy(32, 32 * Direction);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
tbz(TMP1, 63, &AgainInternal256);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbz(TMP1, 63, &AgainInternal128);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
add(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32 - Size);
}
}
Bind(&AgainInternal);
if (Op->IsAtomic) {
if (IsAtomic) {
MemCpyTSO(OpSize, SizeDirection);
}
else {
@@ -2121,8 +2249,8 @@ DEF_OP(MemCpy) {
};
if (DirectionIsInline) {
// If the direction constant is set then the direction is negative.
EmitMemcpy(DirectionConstant ? -1 : 1);
LOGMAN_THROW_AA_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
EmitMemcpy(DirectionConstant);
}
else {
// Emit forward direction memset then backward direction memset.
@@ -2419,6 +2547,47 @@ DEF_OP(CacheLineZero) {
}
}
DEF_OP(Prefetch) {
auto Op = IROp->C<IR::IROp_Prefetch>();
const auto MemReg = GetReg(Op->Addr.ID());
// Access size is only ever handled as 8-byte. Even though it is accesssed as a cacheline.
const auto MemSrc = GenerateMemOperand(8, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
size_t LUT =
(Op->Stream ? 1 : 0) |
((Op->CacheLevel - 1) << 1) |
(Op->ForStore ? 1U << 3 : 0);
constexpr static std::array<ARMEmitter::Prefetch, 14> PrefetchType = {
ARMEmitter::Prefetch::PLDL1KEEP,
ARMEmitter::Prefetch::PLDL1STRM,
ARMEmitter::Prefetch::PLDL2KEEP,
ARMEmitter::Prefetch::PLDL2STRM,
ARMEmitter::Prefetch::PLDL3KEEP,
ARMEmitter::Prefetch::PLDL3STRM,
// Gap of two.
// 0b0'11'0
ARMEmitter::Prefetch::PLDL1STRM,
// 0b0'11'1
ARMEmitter::Prefetch::PLDL1STRM,
ARMEmitter::Prefetch::PSTL1KEEP,
ARMEmitter::Prefetch::PSTL1STRM,
ARMEmitter::Prefetch::PSTL2KEEP,
ARMEmitter::Prefetch::PSTL2STRM,
ARMEmitter::Prefetch::PSTL3KEEP,
ARMEmitter::Prefetch::PSTL3STRM,
};
prfm(PrefetchType[LUT], MemSrc);
}
#undef DEF_OP
}
+1 -1
View File
@@ -1,7 +1,7 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Core/CPUBackend.h>
#include "Interface/Core/CPUBackend.h"
#include <FEXCore/fextl/memory.h>
namespace FEXCore::Context {
+1 -1
View File
@@ -157,7 +157,7 @@ public:
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
// This needs to be taken before reads or writes to L2, L3, CodePages,
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
// may only happen during cross thread invalidation (::Erase).
// All other operations must be done from the owning thread.
File diff suppressed because it is too large. Load diff
+87 -119
View File
@@ -9,7 +9,6 @@
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
@@ -89,10 +88,6 @@ public:
TYPE_LSHRDI,
TYPE_ASHR,
TYPE_ASHRI,
TYPE_ROR,
TYPE_RORI,
TYPE_ROL,
TYPE_ROLI,
TYPE_BEXTR,
TYPE_BLSI,
TYPE_BLSMSK,
@@ -228,6 +223,32 @@ public:
return CanHaveSideEffects;
}
template <typename F>
void ForeachDirection(F&& Routine) {
// Otherwise, prepare to branch.
auto Zero = _Constant(0);
// If the shift is zero, do not touch the flags.
auto ForwardBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
auto BackwardBlock = CreateNewCodeBlockAfter(ForwardBlock);
auto ExitBlock = CreateNewCodeBlockAfter(BackwardBlock);
auto DF = GetRFLAG(X86State::RFLAG_DF_RAW_LOC);
CondJump(DF, Zero, ForwardBlock, BackwardBlock, {COND_EQ});
for (auto D = 0; D < 2; ++D) {
SetCurrentCodeBlock(D ? BackwardBlock : ForwardBlock);
StartNewBlock();
{
Routine(D ? -1 : 1);
Jump(ExitBlock);
}
}
SetCurrentCodeBlock(ExitBlock);
StartNewBlock();
}
OpDispatchBuilder(FEXCore::Context::ContextImpl *ctx);
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
@@ -308,25 +329,22 @@ public:
void CMOVOp(OpcodeArgs);
void CPUIDOp(OpcodeArgs);
void XGetBVOp(OpcodeArgs);
template<bool SHL1Bit>
uint32_t LoadConstantShift(X86Tables::DecodedOp Op, bool Is1Bit);
void SHLOp(OpcodeArgs);
template<bool SHL1Bit>
void SHLImmediateOp(OpcodeArgs);
template<bool SHR1Bit>
void SHROp(OpcodeArgs);
template<bool SHR1Bit>
void SHRImmediateOp(OpcodeArgs);
void SHLDOp(OpcodeArgs);
void SHLDImmediateOp(OpcodeArgs);
void SHRDOp(OpcodeArgs);
void SHRDImmediateOp(OpcodeArgs);
template<bool SHR1Bit>
void ASHROp(OpcodeArgs);
template<bool SHR1Bit>
void ASHRImmediateOp(OpcodeArgs);
template<bool Is1Bit>
void ROROp(OpcodeArgs);
void RORImmediateOp(OpcodeArgs);
template<bool Is1Bit>
void ROLOp(OpcodeArgs);
void ROLImmediateOp(OpcodeArgs);
template<bool Left, bool IsImmediate, bool Is1Bit>
void RotateOp(OpcodeArgs);
void RCROp1Bit(OpcodeArgs);
void RCROp8x1Bit(OpcodeArgs);
void RCROp(OpcodeArgs);
@@ -809,6 +827,7 @@ public:
void FXSaveOp(OpcodeArgs);
void FXRStoreOp(OpcodeArgs);
OrderedNode *XSaveBase(X86Tables::DecodedOp Op);
void XSaveOp(OpcodeArgs);
void PAlignrOp(OpcodeArgs);
@@ -867,6 +886,10 @@ public:
void StoreFenceOrCLFlush(OpcodeArgs);
void CLZeroOp(OpcodeArgs);
void RDTSCPOp(OpcodeArgs);
void RDPIDOp(OpcodeArgs);
template<bool ForStore, bool Stream, uint8_t Level>
void Prefetch(OpcodeArgs);
void PSADBW(OpcodeArgs);
@@ -977,16 +1000,7 @@ private:
}
static bool IsNZCV(unsigned BitOffset) {
switch (BitOffset) {
case FEXCore::X86State::RFLAG_CF_RAW_LOC:
case FEXCore::X86State::RFLAG_ZF_RAW_LOC:
case FEXCore::X86State::RFLAG_SF_RAW_LOC:
case FEXCore::X86State::RFLAG_OF_RAW_LOC:
return true;
default:
return false;
}
return ContainsNZCV(1U << BitOffset);
}
OrderedNode* CachedNZCV{};
@@ -1291,13 +1305,9 @@ private:
}
OrderedNode *GetNZCV() {
if (!CachedNZCV) {
if (!CachedNZCV)
CachedNZCV = _LoadNZCV();
// We don't know what's set
PossiblySetNZCVBits = ~0;
}
return CachedNZCV;
}
@@ -1324,10 +1334,8 @@ private:
}
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
HandleNZ00Write();
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
CachedNZCV = _LoadNZCV();
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
NZCVDirty = false;
}
void InsertNZCV(unsigned BitOffset, OrderedNode *Value, signed FlagOffset, bool MustMask) {
@@ -1401,6 +1409,11 @@ private:
if (ValueOffset || MustMask)
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
// For DF, we need to transform 0/1 into 1/-1
if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
Value = _SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1);
}
_StoreFlag(Value, BitOffset);
}
}
@@ -1413,7 +1426,7 @@ private:
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(Constant << 4));
}
void ZeroMultipleFlags(uint32_t BitMask);
void ZeroPF_AF();
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
switch (BitOffset) {
@@ -1453,11 +1466,32 @@ private:
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
// Recover the sign bit, it is the logical DF value
return _Lshr(OpSize::i64Bit, _LoadDF(), _Constant(63));
} else {
return _LoadFlag(BitOffset);
}
}
// Returns (DF ? -Size : Size)
OrderedNode *LoadDir(const unsigned Size) {
auto Dir = _LoadDF();
auto Shift = FEXCore::ilog2(Size);
if (Shift)
return _Lshl(IR::SizeToOpSize(CTX->GetGPRSize()), Dir, _Constant(Shift));
else
return Dir;
}
// Returns DF ? (X - Size) : (X + Size)
OrderedNode *OffsetByDir(OrderedNode *X, const unsigned Size) {
auto Shift = FEXCore::ilog2(Size);
return _AddShift(OpSize::i64Bit, X, _LoadDF(), ShiftType::LSL, Shift);
}
// Set SSE comparison flags based on the result set by Arm FCMP. This converts
// NZCV from the Arm representation to an eXternal representation that's
// totally not a euphemism for x86 or anything, nuh-uh.
@@ -1597,7 +1631,7 @@ private:
}
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP) const;
OrderedNode *SelectBit(OrderedNode *Cmp, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
OrderedNode *SelectBit(OrderedNode *Cmp, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
OrderedNode *SelectCC(uint8_t OP, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
/**
@@ -1639,13 +1673,13 @@ private:
OrderedNode *Src1;
} OneSource;
// Logical, LSHL, LSHR, ASHR, ROR, ROL
// Logical, LSHL, LSHR, ASHR
struct {
OrderedNode *Src1;
OrderedNode *Src2;
} TwoSource;
// LSHLI, LSHRI, ASHRI, RORI, ROLI
// LSHLI, LSHRI, ASHRI
struct {
OrderedNode *Src1;
uint64_t Imm;
@@ -1694,15 +1728,12 @@ private:
}
template <typename F>
void CalculateFlags_ShiftVariable(OrderedNode *Shift, F&& CalculateFlags) {
// We are the ones calculating the deferred flags. Don't recurse!
InvalidateDeferredFlags();
void Calculate_ShiftVariable(OrderedNode *Shift, F&& Calculate) {
// RCR can call this with constants, so handle that without branching.
uint64_t Const;
if (IsValueConstant(WrapNode(Shift), &Const)) {
if (Const)
CalculateFlags();
Calculate();
return;
}
@@ -1719,7 +1750,7 @@ private:
SetCurrentCodeBlock(SetBlock);
StartNewBlock();
{
CalculateFlags();
Calculate();
Jump(EndBlock);
}
@@ -1728,12 +1759,21 @@ private:
PossiblySetNZCVBits |= OldSetNZCVBits;
}
template <typename F>
void CalculateFlags_ShiftVariable(OrderedNode *Shift, F&& CalculateFlags) {
// We are the ones calculating the deferred flags. Don't recurse!
InvalidateDeferredFlags();
Calculate_ShiftVariable(Shift, CalculateFlags);
}
/**
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
* @{ */
OrderedNode *LoadPFRaw();
OrderedNode *LoadPFRaw(bool Invert);
OrderedNode *LoadAF();
void FixupAF();
void SetAFAndFixup(OrderedNode *AF);
OrderedNode *CalculateAFForDecimal(OrderedNode *A);
void CalculatePF(OrderedNode *Res);
void CalculateAF(OrderedNode *Src1, OrderedNode *Src2);
@@ -1753,10 +1793,6 @@ private:
void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_BEXTR(OrderedNode *Src);
void CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
void CalculateFlags_BLSMSK(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
@@ -1944,78 +1980,6 @@ private:
};
}
void GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
// Doesn't set all the flags, needs to calculate.
CalculateDeferredFlags();
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_ROR,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.TwoSource = {
.Src1 = Src1,
.Src2 = Src2,
},
},
};
}
void GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
// Doesn't set all the flags, needs to calculate.
CalculateDeferredFlags();
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_ROL,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.TwoSource = {
.Src1 = Src1,
.Src2 = Src2,
},
},
};
}
void GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
if (Shift == 0) return;
// Doesn't set all the flags, needs to calculate.
CalculateDeferredFlags();
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_RORI,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.OneSrcImmediate = {
.Src1 = Src1,
.Imm = Shift,
},
},
};
}
void GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
if (Shift == 0) return;
// Doesn't set all the flags, needs to calculate.
CalculateDeferredFlags();
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_ROLI,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.OneSrcImmediate = {
.Src1 = Src1,
.Imm = Shift,
},
}
};
}
void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_BEXTR,
@@ -2145,6 +2109,10 @@ private:
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
OrderedNode* Prefetch(bool ForStore, bool Stream, uint8_t CacheLevel, OrderedNode *ssa0) {
return _Prefetch(ForStore, Stream, CacheLevel, ssa0, Invalid(), MEM_OFFSET_SXTX, 1);
}
void InstallHostSpecificOpcodeHandlers();
///< Segment telemetry tracking
@@ -13,7 +13,6 @@ $end_info$
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/IR/IR.h>
#include <array>
#include <cstdint>
@@ -27,7 +26,7 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
FEXCore::X86State::RFLAG_SF_RAW_LOC,
FEXCore::X86State::RFLAG_TF_LOC,
FEXCore::X86State::RFLAG_IF_LOC,
FEXCore::X86State::RFLAG_DF_LOC,
FEXCore::X86State::RFLAG_DF_RAW_LOC,
FEXCore::X86State::RFLAG_OF_RAW_LOC,
FEXCore::X86State::RFLAG_IOPL_LOC,
FEXCore::X86State::RFLAG_NT_LOC,
@@ -39,61 +38,10 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
FEXCore::X86State::RFLAG_ID_LOC,
};
void OpDispatchBuilder::ZeroMultipleFlags(uint32_t FlagsMask) {
auto ZeroConst = _Constant(0);
if (ContainsNZCV(FlagsMask)) {
// NZCV is stored packed together.
// It's more optimal to zero NZCV with move+bic instead of multiple bics.
auto NZCVFlagsMask = FlagsMask & FullNZCVMask;
if (NZCVFlagsMask == FullNZCVMask) {
ZeroNZCV();
}
else {
const auto IndexMask = NZCVIndexMask(FlagsMask);
if (std::popcount(NZCVFlagsMask) == 1) {
// It's more optimal to store only one here.
for (size_t i = 0; NZCVFlagsMask && i < FlagOffsets.size(); ++i) {
const auto FlagOffset = FlagOffsets[i];
const auto FlagMask = 1U << FlagOffset;
if (!(FlagMask & NZCVFlagsMask)) {
continue;
}
SetRFLAG(ZeroConst, FlagOffset);
NZCVFlagsMask &= ~(FlagMask);
}
}
else {
auto IndexMaskConstant = _Constant(IndexMask);
auto NewNZCV = _Andn(OpSize::i64Bit, GetNZCV(), IndexMaskConstant);
SetNZCV(NewNZCV);
}
// Unset the possibly set bits.
PossiblySetNZCVBits &= ~IndexMask;
}
// Handled NZCV, so remove it from the mask.
FlagsMask &= ~FullNZCVMask;
}
void OpDispatchBuilder::ZeroPF_AF() {
// PF is stored inverted, so invert it when we zero.
if (FlagsMask & (1u << X86State::RFLAG_PF_RAW_LOC)) {
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Constant(1));
FlagsMask &= ~(1u << X86State::RFLAG_PF_RAW_LOC);
}
// Handle remaining masks.
for (size_t i = 0; FlagsMask && i < FlagOffsets.size(); ++i) {
const auto FlagOffset = FlagOffsets[i];
const auto FlagMask = 1U << FlagOffset;
if (!(FlagMask & FlagsMask)) {
continue;
}
SetRFLAG(ZeroConst, FlagOffset);
FlagsMask &= ~(FlagMask);
}
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Constant(1));
SetAF(0);
}
void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
@@ -180,7 +128,7 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
// instead.
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_RAW_LOC)) {
// Set every bit except the bottommost.
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(), _Constant(~1ull));
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false), _Constant(~1ull));
// Rotate the bottom bit to the appropriate location for PF, so we get
// something like 111P1111. Then invert that to get 000p0000. Then OR that
@@ -238,18 +186,21 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNo
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
}
OrderedNode *OpDispatchBuilder::LoadPFRaw() {
OrderedNode *OpDispatchBuilder::LoadPFRaw(bool Invert) {
// Read the stored byte. This is the original result (up to 64-bits), it needs
// parity calculated.
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
// Cast the input to a 32-bit FPR. Logically we only need 8-bit, but that would
// generate unwanted an ubfx instruction. VPopcount will ignore the upper bits anyway.
auto InputFPR = _VCastFromGPR(4, 4, Result);
// Cascade to calculate parity of bottom 8-bits to bottom bit.
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 4);
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 2);
// Calculate the popcount.
auto Count = _VPopcount(1, 1, InputFPR);
return _VExtractToGPR(8, 1, Count, 0);
if (Invert)
Result = _XornShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
else
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
return Result;
}
OrderedNode *OpDispatchBuilder::LoadAF() {
@@ -277,6 +228,19 @@ void OpDispatchBuilder::FixupAF() {
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
}
void OpDispatchBuilder::SetAFAndFixup(OrderedNode *AF) {
// We have a value of AF, we shift into AF[4]. We need to fixup AF[4] so that
// we get the right value when we XOR in PF[4] later. The easiest solution is
// to XOR by PF[4], since:
//
// (AF[4] ^ PF[4]) ^ PF[4] = AF[4]
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
OrderedNode *XorRes = _XorShift(OpSize::i32Bit, PFRaw, AF, ShiftType::LSL, 4);
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
}
void OpDispatchBuilder::CalculatePF(OrderedNode *Res) {
// Calculation is entirely deferred until load, just store the 8-bit result.
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Res);
@@ -388,34 +352,6 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
break;
case FlagsGenerationType::TYPE_ROR:
CalculateFlags_RotateRight(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.TwoSource.Src1,
CurrentDeferredFlags.Sources.TwoSource.Src2);
break;
case FlagsGenerationType::TYPE_RORI:
CalculateFlags_RotateRightImmediate(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
break;
case FlagsGenerationType::TYPE_ROL:
CalculateFlags_RotateLeft(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.TwoSource.Src1,
CurrentDeferredFlags.Sources.TwoSource.Src2);
break;
case FlagsGenerationType::TYPE_ROLI:
CalculateFlags_RotateLeftImmediate(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
break;
case FlagsGenerationType::TYPE_BEXTR:
CalculateFlags_BEXTR(CurrentDeferredFlags.Res);
break;
@@ -822,107 +758,6 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
}
}
void OpDispatchBuilder::CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
auto SizeBits = SrcSize * 8;
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
// OF is set to the XOR of the new CF bit and the most significant bit of the result
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, true);
});
}
void OpDispatchBuilder::CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
// Extract the last bit shifted in to CF
//auto Size = _Constant(GetSrcSize(Res) * 8);
//auto ShiftAmt = _Sub(OpSize::i64Bit, Size, Src2);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
// OF is the LSB and MSB XOR'd together.
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
});
}
void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
if (Shift == 0) return;
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
}
// OF
{
if (Shift == 1) {
// OF is the top two MSBs XOR'd together
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, 1);
}
}
}
void OpDispatchBuilder::CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
if (Shift == 0) return;
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
}
// OF
{
if (Shift == 1) {
// OF is the LSB and MSB XOR'd together.
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
}
}
}
void OpDispatchBuilder::CalculateFlags_BEXTR(OrderedNode *Src) {
// ZF is set properly. CF and OF are defined as being set to zero. SF, PF, and
// AF are undefined.
@@ -982,9 +817,7 @@ void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode *Result) {
// is in the range [0, 63]. In particular, it is always positive. So a
// combined NZ test will correctly zero SF/CF/OF while setting ZF.
SetNZ_ZeroCV(OpSize::i32Bit, Result);
ZeroMultipleFlags((1U << X86State::RFLAG_AF_RAW_LOC) |
(1U << X86State::RFLAG_PF_RAW_LOC));
ZeroPF_AF();
}
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
@@ -1010,15 +843,10 @@ void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, OrderedNode *Result
void OpDispatchBuilder::CalculateFlags_RDRAND(OrderedNode *Src) {
// OF, SF, ZF, AF, PF all zero
ZeroNZCV();
ZeroPF_AF();
// CF is set to the incoming source
uint32_t FlagsMaskToZero =
FullNZCVMask |
(1U << X86State::RFLAG_AF_RAW_LOC) |
(1U << X86State::RFLAG_PF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src);
}
@@ -13,7 +13,6 @@ $end_info$
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
@@ -1104,7 +1103,7 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
const auto ExtractSize = Is256Bit ? 4 : 2;
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *VMask = _VDupFromGPR(SrcSize, 8, _Constant(0x80'40'20'10'08'04'02'01ULL));
OrderedNode *VMask = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_MOVMASKB);
auto VCMP = _VCMPLTZ(SrcSize, 1, Src);
auto VAnd = _VAnd(SrcSize, 1, VCMP, VMask);
@@ -3001,16 +3000,15 @@ void OpDispatchBuilder::XSaveOp(OpcodeArgs) {
XSaveOpImpl(Op);
}
void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
const auto XSaveBase = [this, Op] {
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
return AppendSegmentOffset(Mem, Op->Flags);
};
OrderedNode *OpDispatchBuilder::XSaveBase(X86Tables::DecodedOp Op) {
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
return AppendSegmentOffset(Mem, Op->Flags);
}
void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
// NOTE: Mask should be EAX and EDX concatenated, but we only need to test
// for features that are in the lower 32 bits, so EAX only is sufficient.
OrderedNode *Mask = LoadGPRRegister(X86State::REG_RAX);
OrderedNode *Base = XSaveBase();
const auto OpSize = IR::SizeToOpSize(CTX->GetGPRSize());
const auto StoreIfFlagSet = [&](uint32_t BitIndex, auto fn, uint32_t FieldSize = 1){
@@ -3034,25 +3032,26 @@ void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
// x87
{
StoreIfFlagSet(0, [this, Op, Base] { SaveX87State(Op, Base); });
StoreIfFlagSet(0, [this, Op] { SaveX87State(Op, XSaveBase(Op)); });
}
// SSE
{
StoreIfFlagSet(1, [this, Base] { SaveSSEState(Base); });
StoreIfFlagSet(1, [this, Op] { SaveSSEState(XSaveBase(Op)); });
}
// AVX
if (CTX->HostFeatures.SupportsAVX)
{
StoreIfFlagSet(2, [this, Base] { SaveAVXState(Base); });
StoreIfFlagSet(2, [this, Op] { SaveAVXState(XSaveBase(Op)); });
}
// We need to save MXCSR and MXCSR_MASK if either SSE or AVX are requested to be saved
{
StoreIfFlagSet(1, [this, Base] { SaveMXCSRState(Base); }, 2);
StoreIfFlagSet(1, [this, Op] { SaveMXCSRState(XSaveBase(Op)); }, 2);
}
// Update XSTATE_BV region of the XSAVE header
{
OrderedNode *Base = XSaveBase(Op);
OrderedNode *HeaderOffset = _Add(OpSize, Base, _Constant(512));
// NOTE: We currently only support the first 3 bits (x87, SSE, and AVX)
@@ -3210,14 +3209,11 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
const auto OpSize = IR::SizeToOpSize(CTX->GetGPRSize());
const auto XSaveBase = [this, Op] {
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
return AppendSegmentOffset(Mem, Op->Flags);
};
// Set up base address for the XSAVE region to restore from, and also read the
// XSTATE_BV bit flags out of the XSTATE header.
OrderedNode *Base = XSaveBase();
//
// Note: we rematerialize Base in each block to avoid crossblock liveness.
OrderedNode *Base = XSaveBase(Op);
OrderedNode *Mask = _LoadMem(GPRClass, 8, _Add(OpSize, Base, _Constant(512)), 8);
// If a bit in our XSTATE_BV is set, then we restore from that region of the XSAVE area,
@@ -3253,27 +3249,28 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
// x87
{
RestoreIfFlagSetOrDefault(0,
[this, Base] { RestoreX87State(Base); },
[this, Op] { RestoreX87State(XSaveBase(Op)); },
[this, Op] { DefaultX87State(Op); });
}
// SSE
{
RestoreIfFlagSetOrDefault(1,
[this, Base] { RestoreSSEState(Base); },
[this, Op] { RestoreSSEState(XSaveBase(Op)); },
[this] { DefaultSSEState(); });
}
// AVX
if (CTX->HostFeatures.SupportsAVX)
{
RestoreIfFlagSetOrDefault(2,
[this, Base] { RestoreAVXState(Base); },
[this, Op] { RestoreAVXState(XSaveBase(Op)); },
[this] { DefaultAVXState(); });
}
{
// We need to restore the MXCSR if either SSE or AVX are requested to be saved
RestoreIfFlagSetOrDefault(1,
[this, Base, OpSize] {
[this, Op, OpSize] {
OrderedNode *Base = XSaveBase(Op);
OrderedNode *MXCSRLocation = _Add(OpSize, Base, _Constant(24));
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
RestoreMXCSRState(MXCSR);
@@ -4597,11 +4594,7 @@ void OpDispatchBuilder::PTestOp(OpcodeArgs) {
SetNZ_ZeroCV(32, Test1);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Test2);
uint32_t FlagsMaskToZero =
(1U << X86State::RFLAG_PF_RAW_LOC) |
(1U << X86State::RFLAG_AF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
ZeroPF_AF();
}
void OpDispatchBuilder::VTESTOpImpl(OpcodeArgs, size_t ElementSize) {
@@ -4638,8 +4631,7 @@ void OpDispatchBuilder::VTESTOpImpl(OpcodeArgs, size_t ElementSize) {
SetNZ_ZeroCV(32, AndGPR);
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFResult);
ZeroMultipleFlags((1U << X86State::RFLAG_PF_RAW_LOC) |
(1U << X86State::RFLAG_AF_RAW_LOC));
ZeroPF_AF();
}
template <size_t ElementSize>
@@ -5571,11 +5563,7 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(GetFlagBit(18));
SetRFLAG<X86State::RFLAG_OF_RAW_LOC>(GetFlagBit(19));
uint32_t FlagsMaskToZero =
(1U << X86State::RFLAG_PF_RAW_LOC) |
(1U << X86State::RFLAG_AF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
ZeroPF_AF();
}
void OpDispatchBuilder::VPCMPESTRIOp(OpcodeArgs) {
@@ -162,7 +162,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
{OPD(TYPE_GROUP_9, PF_F3, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"RDPID", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"RDPID", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
@@ -6,13 +6,13 @@ tags: glue|thunks
$end_info$
*/
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
+2 -1
View File
@@ -7,7 +7,8 @@ $end_info$
#pragma once
#include <FEXCore/IR/IR.h>
#include "Interface/IR/IR.h"
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/vector.h>
+2 -2
View File
@@ -2,9 +2,9 @@
#include "FEXHeaderUtils/Filesystem.h"
#include "Interface/Context/Context.h"
#include "Interface/IR/AOTIR.h"
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/fextl/fmt.h>
+2 -1
View File
@@ -1,7 +1,8 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "FEXCore/IR/RegisterAllocationData.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/string.h>
+617
View File
@@ -9,11 +9,628 @@
namespace FEXCore::IR {
class OrderedNode;
class RegisterAllocationPass;
class RegisterAllocationData;
/**
* @brief The IROp_Header is an dynamically sized array
* At the end it contains a uint8_t for the number of arguments that Op has
* Then there is an unsized array of NodeWrapper arguments for the number of arguments this op has
* The op structures that are including the header must ensure that they pad themselves correctly to the number of arguments used
*/
struct IROp_Header;
/**
* @brief Represents the ID of a given IR node.
*
* Intended to provide strong typing from other integer values
* to prevent passing incorrect values to certain API functions.
*/
struct NodeID final {
using value_type = uint32_t;
constexpr NodeID() noexcept = default;
constexpr explicit NodeID(value_type Value_) noexcept : Value{Value_} {}
constexpr NodeID(const NodeID&) noexcept = default;
constexpr NodeID& operator=(const NodeID&) noexcept = default;
constexpr NodeID(NodeID&&) noexcept = default;
constexpr NodeID& operator=(NodeID&&) noexcept = default;
[[nodiscard]] constexpr bool IsValid() const noexcept {
return Value != 0;
}
[[nodiscard]] constexpr bool IsInvalid() const noexcept {
return !IsValid();
}
constexpr void Invalidate() noexcept {
Value = 0;
}
[[nodiscard]] friend constexpr bool operator==(NodeID, NodeID) noexcept = default;
[[nodiscard]] friend constexpr bool operator<(NodeID lhs, NodeID rhs) noexcept {
return lhs.Value < rhs.Value;
}
[[nodiscard]] friend constexpr bool operator>(NodeID lhs, NodeID rhs) noexcept {
return operator<(rhs, lhs);
}
[[nodiscard]] friend constexpr bool operator<=(NodeID lhs, NodeID rhs) noexcept {
return !operator>(lhs, rhs);
}
[[nodiscard]] friend constexpr bool operator>=(NodeID lhs, NodeID rhs) noexcept {
return !operator<(lhs, rhs);
}
friend std::ostream& operator<<(std::ostream& out, NodeID ID) {
out << ID.Value;
return out;
}
friend std::istream& operator>>(std::istream& in, NodeID& ID) {
in >> ID.Value;
return in;
}
value_type Value{};
};
/**
* @brief This is a very simple wrapper for our node pointers
* You probably don't want to use this directly
* Use OpNodeWrapper and OrderedNodeWrapper types below instead
*
* This is necessary to allow two things
* - Reduce memory usage by having the pointer be an 32bit offset rather than the whole 64bit pointer
* - Actually use an offset from a base so we aren't storing pointers for everything
* - Makes IR list copying be as cheap as a memcpy
* Downsides
* - The IR nodes have to be allocated out of a linear array of memory
* - We currently only allow a 32bit offset, so *only* 4 million nodes per list
* - We have to have the base offset live somewhere else
* - Has to be POD and trivially copyable
* - Makes every real node access turn in to a [Base + Offset] access
* - Can be confusing if you're mixing OpNodeWrapper and OrderedNodeWrapper usage
*/
template<typename Type>
struct NodeWrapperBase final {
// On x86-64 using a uint64_t type is more efficient since RIP addressing gives you [<Base> + <Index> + <imm offset>]
// On AArch64 using uint32_t is just more memory efficient. 32bit or 64bit offset doesn't matter
// We use uint32_t to be more memory efficient (Cuts our node list size in half)
using NodeOffsetType = uint32_t;
NodeOffsetType NodeOffset;
explicit NodeWrapperBase() = default;
[[nodiscard]] static NodeWrapperBase WrapOffset(NodeOffsetType Offset) {
NodeWrapperBase Wrapped;
Wrapped.NodeOffset = Offset;
return Wrapped;
}
[[nodiscard]] static NodeWrapperBase WrapPtr(uintptr_t Base, uintptr_t Value) {
NodeWrapperBase Wrapped;
Wrapped.SetOffset(Base, Value);
return Wrapped;
}
[[nodiscard]] static void *UnwrapNode(uintptr_t Base, NodeWrapperBase Node) {
return Node.GetNode(Base);
}
[[nodiscard]] NodeID ID() const;
[[nodiscard]] bool IsInvalid() const { return NodeOffset == 0; }
[[nodiscard]] Type *GetNode(uintptr_t Base) {
return reinterpret_cast<Type*>(Base + NodeOffset);
}
[[nodiscard]] const Type *GetNode(uintptr_t Base) const {
return reinterpret_cast<const Type*>(Base + NodeOffset);
}
void SetOffset(uintptr_t Base, uintptr_t Value) { NodeOffset = Value - Base; }
[[nodiscard]] friend constexpr bool operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
};
static_assert(std::is_trivial_v<NodeWrapperBase<OrderedNode>>);
static_assert(sizeof(NodeWrapperBase<OrderedNode>) == sizeof(uint32_t));
using OpNodeWrapper = NodeWrapperBase<IROp_Header>;
using OrderedNodeWrapper = NodeWrapperBase<OrderedNode>;
struct OrderedNodeHeader {
OpNodeWrapper Value;
OrderedNodeWrapper Next;
OrderedNodeWrapper Previous;
};
static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3);
/**
* @brief This is a node in our IR representation
* Is a doubly linked list node that lives in a representation of a linearly allocated node list
* The links in the nodes can live in a list independent of the data IR data
*
* ex.
* Region1 : ... <-> <OrderedNode> <-> <OrderedNode> <-> ...
* | *<Value> |
* v v
* Region2 : <IROp>..<IROp>..<IROp>..<IROp>
*
* In this example the OrderedNodes are allocated in one linear memory region (Not necessarily contiguous with one another linking)
* The second region is contiguous but they don't have any relationship with one another directly
*/
class OrderedNode final {
friend class NodeWrapperIterator;
friend class OrderedList;
public:
// These three values are laid out very specifically to make it fast to access the NodeWrappers specifically
OrderedNodeHeader Header;
uint32_t NumUses;
using value_type = OrderedNodeWrapper;
OrderedNode() = default;
/**
* @brief Appends a node to this current node
*
* Before. <Prev> <-> <Current> <-> <Next>
* After. <Prev> <-> <Current> <-> <Node> <-> Next
*
* @return Pointer to the node being added
*/
value_type append(uintptr_t Base, value_type Node) {
// Set Next Node's Previous to incoming node
SetPrevious(Base, Header.Next, Node);
// Set Incoming node's links to this node's links
SetPrevious(Base, Node, Wrapped(Base));
SetNext(Base, Node, Header.Next);
// Set this node's next to the incoming node
SetNext(Base, Wrapped(Base), Node);
// Return the node we are appending
return Node;
}
OrderedNode *append(uintptr_t Base, OrderedNode *Node) {
value_type WNode = Node->Wrapped(Base);
// Set Next Node's Previous to incoming node
SetPrevious(Base, Header.Next, WNode);
// Set Incoming node's links to this node's links
SetPrevious(Base, WNode, Wrapped(Base));
SetNext(Base, WNode, Header.Next);
// Set this node's next to the incoming node
SetNext(Base, Wrapped(Base), WNode);
// Return the node we are appending
return Node;
}
/**
* @brief Prepends a node to the current node
* Before. <Prev> <-> <Current> <-> <Next>
* After. <Prev> <-> <Node> <-> <Current> <-> Next
*
* @return Pointer to the node being added
*/
value_type prepend(uintptr_t Base, value_type Node) {
// Set the previous node's next to the incoming node
SetNext(Base, Header.Previous, Node);
// Set the incoming node's links
SetPrevious(Base, Node, Header.Previous);
SetNext(Base, Node, Wrapped(Base));
// Set the current node's link
SetPrevious(Base, Wrapped(Base), Node);
// Return the node we are prepending
return Node;
}
OrderedNode *prepend(uintptr_t Base, OrderedNode *Node) {
value_type WNode = Node->Wrapped(Base);
// Set the previous node's next to the incoming node
SetNext(Base, Header.Previous, WNode);
// Set the incoming node's links
SetPrevious(Base, WNode, Header.Previous);
SetNext(Base, WNode, Wrapped(Base));
// Set the current node's link
SetPrevious(Base, Wrapped(Base), WNode);
// Return the node we are prepending
return Node;
}
/**
* @brief Gets the remaining size of the blocks from this point onward
*
* Doesn't find the head of the list
*
*/
[[nodiscard]] size_t size(uintptr_t Base) const {
size_t Size = 1;
// Walk the list forward until we hit a sentinel
value_type Current = Header.Next;
while (Current.NodeOffset != 0) {
++Size;
OrderedNode *RealNode = Current.GetNode(Base);
Current = RealNode->Header.Next;
}
return Size;
}
void Unlink(uintptr_t Base) {
// This removes the node from the list. Orphaning it
// Before: <Previous> <-> <Current> <-> <Next>
// After: <Previous <-> <Next>
SetNext(Base, Header.Previous, Header.Next);
SetPrevious(Base, Header.Next, Header.Previous);
}
[[nodiscard]] IROp_Header const* Op(uintptr_t Base) const {
return Header.Value.GetNode(Base);
}
[[nodiscard]] IROp_Header *Op(uintptr_t Base) {
return Header.Value.GetNode(Base);
}
[[nodiscard]] uint32_t GetUses() const { return NumUses; }
void AddUse() { ++NumUses; }
void RemoveUse() { --NumUses; }
[[nodiscard]] value_type Wrapped(uintptr_t Base) const {
value_type Tmp;
Tmp.SetOffset(Base, reinterpret_cast<uintptr_t>(this));
return Tmp;
}
private:
[[nodiscard]] value_type WrappedOffset(uint32_t Offset) const {
value_type Tmp;
Tmp.NodeOffset = Offset;
return Tmp;
}
static void SetPrevious(uintptr_t Base, value_type Node, value_type New) {
OrderedNode *RealNode = Node.GetNode(Base);
RealNode->Header.Previous = New;
}
static void SetNext(uintptr_t Base, value_type Node, value_type New) {
OrderedNode *RealNode = Node.GetNode(Base);
RealNode->Header.Next = New;
}
void SetUses(uint32_t Uses) { NumUses = Uses; }
};
static_assert(std::is_trivial_v<OrderedNode>);
static_assert(std::is_trivially_copyable_v<OrderedNode>);
static_assert(offsetof(OrderedNode, Header) == 0);
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
struct RegisterClassType final {
using value_type = uint32_t;
value_type Val;
[[nodiscard]] constexpr operator value_type() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
};
struct CondClassType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const CondClassType&, const CondClassType&) = default;
};
struct MemOffsetType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const MemOffsetType&, const MemOffsetType&) = default;
};
struct TypeDefinition final {
uint16_t Val;
[[nodiscard]] constexpr operator uint16_t() const {
return Val;
}
[[nodiscard]] static constexpr TypeDefinition Create(uint8_t Bytes) {
TypeDefinition Type{};
Type.Val = Bytes << 8;
return Type;
}
[[nodiscard]] static constexpr TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
TypeDefinition Type{};
Type.Val = (Bytes << 8) | (Elements & 255);
return Type;
}
[[nodiscard]] constexpr uint8_t Bytes() const {
return Val >> 8;
}
[[nodiscard]] constexpr uint8_t Elements() const {
return Val & 255;
}
[[nodiscard]] friend constexpr bool operator==(const TypeDefinition&, const TypeDefinition&) = default;
};
static_assert(std::is_trivial_v<TypeDefinition>);
struct FenceType final {
using value_type = uint8_t;
value_type Val;
[[nodiscard]] constexpr operator value_type() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
};
struct RoundType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const RoundType&, const RoundType&) = default;
};
class NodeIterator;
/* This iterator can be used to step though nodes.
* Due to how our IR is laid out, this can be used to either step
* though the CodeBlocks or though the code within a single block.
*/
class NodeIterator {
public:
using value_type = std::tuple<OrderedNode*, IROp_Header*>;
using size_type = std::size_t;
using difference_type = std::ptrdiff_t;
using reference = value_type&;
using const_reference = const value_type&;
using pointer = value_type*;
using const_pointer = const value_type*;
using iterator = NodeIterator;
using const_iterator = const NodeIterator;
using reverse_iterator = iterator;
using const_reverse_iterator = const_iterator;
using iterator_category = std::bidirectional_iterator_tag;
NodeIterator(uintptr_t Base, uintptr_t IRBase) : BaseList {Base}, IRList{ IRBase } {}
explicit NodeIterator(uintptr_t Base, uintptr_t IRBase, OrderedNodeWrapper Ptr) : BaseList {Base}, IRList{ IRBase }, Node {Ptr} {}
[[nodiscard]] bool operator==(const NodeIterator &rhs) const {
return Node.NodeOffset == rhs.Node.NodeOffset;
}
[[nodiscard]] bool operator!=(const NodeIterator &rhs) const {
return !operator==(rhs);
}
NodeIterator operator++() {
OrderedNodeHeader *RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
Node = RealNode->Next;
return *this;
}
NodeIterator operator--() {
OrderedNodeHeader *RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
Node = RealNode->Previous;
return *this;
}
[[nodiscard]] value_type operator*() {
OrderedNode *RealNode = Node.GetNode(BaseList);
return { RealNode, RealNode->Op(IRList) };
}
[[nodiscard]] value_type operator()() {
OrderedNode *RealNode = Node.GetNode(BaseList);
return { RealNode, RealNode->Op(IRList) };
}
[[nodiscard]] NodeID ID() const {
return Node.ID();
}
[[nodiscard]] static NodeIterator Invalid() {
return NodeIterator(0, 0);
}
protected:
uintptr_t BaseList{};
uintptr_t IRList{};
OrderedNodeWrapper Node{};
};
// This must directly match bytes to the named opsize.
// Implicit sized IR operations does math to get between sizes.
enum OpSize : uint8_t {
i8Bit = 1,
i16Bit = 2,
i32Bit = 4,
i64Bit = 8,
i128Bit = 16,
i256Bit = 32,
};
enum class FloatCompareOp : uint8_t {
EQ = 0,
LT,
LE,
UNO,
NEQ,
ORD,
};
enum class ShiftType : uint8_t {
LSL = 0,
LSR,
ASR,
ROR,
};
// Converts a size stored as an integer in to an OpSize enum.
// This is a nop operation and will be eliminated by the compiler.
static inline OpSize SizeToOpSize(uint8_t Size) {
switch (Size) {
case 1: return OpSize::i8Bit;
case 2: return OpSize::i16Bit;
case 4: return OpSize::i32Bit;
case 8: return OpSize::i64Bit;
case 16: return OpSize::i128Bit;
case 32: return OpSize::i256Bit;
default: FEX_UNREACHABLE;
}
}
#define IROP_ENUM
#define IROP_STRUCTS
#define IROP_SIZES
#define IROP_REG_CLASSES
#include <FEXCore/IR/IRDefines.inc>
/* This iterator can be used to step though every single node in a multi-block in SSA order.
*
* Iterates in the order of:
*
* end <-- CodeBlockA <--> BlockAInst1 <--> BlockAInst2 <--> CodeBlockB <--> BlockBInst1 <--> BlockBInst2 --> end
*/
class AllNodesIterator : public NodeIterator {
public:
AllNodesIterator(uintptr_t Base, uintptr_t IRBase) : NodeIterator(Base, IRBase) {}
explicit AllNodesIterator(uintptr_t Base, uintptr_t IRBase, OrderedNodeWrapper Ptr) : NodeIterator(Base, IRBase, Ptr) {}
AllNodesIterator(NodeIterator other) : NodeIterator(other) {} // Allow NodeIterator to be upgraded
AllNodesIterator operator++() {
OrderedNodeHeader *RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
auto IROp = Node.GetNode(BaseList)->Op(IRList);
// If this is the last node of a codeblock, we need to continue to the next block
if (IROp->Op == OP_ENDBLOCK) {
auto EndBlock = IROp->C<IROp_EndBlock>();
auto CurrentBlock = EndBlock->BlockHeader.GetNode(BaseList);
Node = CurrentBlock->Header.Next;
} else if (IROp->Op == OP_CODEBLOCK) {
auto CodeBlock = IROp->C<IROp_CodeBlock>();
Node = CodeBlock->Begin;
} else {
Node = RealNode->Next;
}
return *this;
}
AllNodesIterator operator--() {
auto IROp = Node.GetNode(BaseList)->Op(IRList);
if (IROp->Op == OP_BEGINBLOCK) {
auto BeginBlock = IROp->C<IROp_EndBlock>();
Node = BeginBlock->BlockHeader;
} else if (IROp->Op == OP_CODEBLOCK) {
auto PrevBlockWrapper = Node.GetNode(BaseList)->Header.Previous;
auto PrevCodeBlock = PrevBlockWrapper.GetNode(BaseList)->Op(IRList)->C<IROp_CodeBlock>();
Node = PrevCodeBlock->Last;
} else {
Node = Node.GetNode(BaseList)->Header.Previous;
}
return *this;
}
[[nodiscard]] static AllNodesIterator Invalid() {
return AllNodesIterator(0, 0);
}
};
class IRListView;
class IREmitter;
template<typename Type>
inline NodeID NodeWrapperBase<Type>::ID() const {
return NodeID(NodeOffset / sizeof(IR::OrderedNode));
}
bool IsFragmentExit(FEXCore::IR::IROps Op);
bool IsBlockExit(FEXCore::IR::IROps Op);
void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData);
fextl::unique_ptr<IREmitter> Parse(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator, fextl::stringstream &MapsStream);
}
template <>
struct std::hash<FEXCore::IR::NodeID> {
size_t operator()(const FEXCore::IR::NodeID& ID) const noexcept {
return std::hash<FEXCore::IR::NodeID::value_type>{}(ID.Value);
}
};
template <>
struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID::value_type> {
using Base = fmt::formatter<FEXCore::IR::NodeID::value_type>;
// Pass-through the underlying value, so IDs can
// be formatted like any integral value.
template <typename FormatContext>
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
return Base::format(ID.Value, ctx);
}
};
template <>
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
template <typename FormatContext>
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
return Base::format(Class.Val, ctx);
}
};
template <>
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
template <typename FormatContext>
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
return Base::format(Fence.Val, ctx);
}
};
template <>
struct fmt::formatter<FEXCore::IR::OpSize> : fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>> {
using Base = fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>>;
template <typename FormatContext>
auto format(const FEXCore::IR::OpSize& OpSize, FormatContext& ctx) const {
return Base::format(FEXCore::ToUnderlying(OpSize), ctx);
}
};
+36 -1
View File
@@ -456,6 +456,13 @@
"DestSize": "4"
},
"GPR = LoadDF": {
"Desc": ["Loads the decimal flag from the context object in -1/1",
"representation for easy consumption"
],
"DestSize": "8"
},
"GPR = LoadFlag u32:$Flag": {
"Desc": ["Loads an x86-64 flag from the context object",
"Specialized to allow flexible implementation of flag handling"
@@ -596,6 +603,15 @@
"Ensures the memory operations are globally visible"
],
"HasSideEffects": true
},
"Prefetch i1:$ForStore, i1:$Stream, i8:$CacheLevel, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
"Desc": ["Does a cacheline prefetch operation"
],
"EmitValidation": [
"_CacheLevel > 0 && _CacheLevel < 4"
],
"HasSideEffects": true,
"DestSize": "8"
}
},
"Atomic": {
@@ -626,7 +642,6 @@
],
"HasDest": true,
"DestSize": "Size",
"ImplicitFlagClobber": true,
"NumElements": "2",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
@@ -1020,6 +1035,14 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"CondSubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2, CondClass:$Cond, u8:$FalseNZCV": {
"Desc": ["If condition is true, set NZCV per difference of GPRs, else force NZCV to a constant."],
"HasSideEffects": true,
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = AdcWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Adds and set NZCV for the sum of two GPRs and carry-in given as NZCV"],
"HasSideEffects": true,
@@ -1079,6 +1102,11 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"CmpPairZ OpSize:#Size, GPRPair:$Src1, GPRPair:$Src2": {
"Desc": ["Compares register pairs and sets Z accordingly, preserving N/Z/V.",
"This accelerates cmpxchg."],
"HasSideEffects": true
},
"SubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the difference of two GPRs. ",
"Carry flag uses arm64 definition, inverted x86.",
@@ -1133,6 +1161,13 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = XornShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
"Desc": [ "Integer binary exclusive or not with shifted register"],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = And OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer binary and"
],
+3 -2
View File
@@ -6,9 +6,10 @@ tags: ir|dumper
$end_info$
*/
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/fextl/sstream.h>
#include <algorithm>
@@ -9,7 +9,6 @@ $end_info$
#include "Interface/IR/IREmitter.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
+4 -1
View File
@@ -1,7 +1,10 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Interface/IR/IR.h"
#include "Interface/IR/IntrusiveIRList.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
+1 -1
View File
@@ -7,8 +7,8 @@ $end_info$
*/
#include "Interface/IR/IREmitter.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/StringUtils.h>
#include <FEXCore/fextl/sstream.h>
@@ -1,7 +1,8 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "FEXCore/IR/IR.h"
#include "Interface/IR/IR.h"
#include <FEXCore/Core/Context.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/LogManager.h>
+2 -2
View File
@@ -80,7 +80,8 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool Inli
InsertPass(CreateDeadStoreElimination(ctx->HostFeatures.SupportsAVX));
InsertPass(CreatePassDeadCodeElimination());
InsertPass(CreateConstProp(InlineConstants, ctx->HostFeatures.SupportsTSOImm9));
InsertPass(CreateConstProp(
InlineConstants, ctx->HostFeatures.SupportsTSOImm9, Is64BitMode()));
InsertPass(CreateDeadFlagCalculationEliminination());
@@ -121,5 +122,4 @@ bool PassManager::Run(IREmitter *IREmit) {
return Changed;
}
}
+5 -3
View File
@@ -16,15 +16,17 @@ class Pass;
class RegisterAllocationPass;
class RegisterAllocationData;
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants, bool SupportsTSOImm9);
fextl::unique_ptr<FEXCore::IR::Pass>
CreateConstProp(bool InlineConstants, bool SupportsTSOImm9, bool Is64BitMode);
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX);
fextl::unique_ptr<FEXCore::IR::Pass> CreateInlineCallOptimization(const FEXCore::CPUIDEmu* CPUID);
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination(bool SupportsAVX);
fextl::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination();
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass,
bool SupportsAVX);
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass>
CreateRegisterAllocationPass(FEXCore::IR::Pass *CompactionPass,
bool SupportsAVX);
fextl::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
namespace Validation {
+221 -116
View File
@@ -17,7 +17,6 @@ $end_info$
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/map.h>
@@ -27,6 +26,7 @@ $end_info$
#include <bit>
#include <cstdint>
#include <memory>
#include <optional>
#include <string.h>
#include <tuple>
#include <utility>
@@ -90,58 +90,132 @@ static bool IsTSOImm9(uint64_t imm) {
}
}
static std::tuple<MemOffsetType, uint8_t, OrderedNode*, OrderedNode*> MemExtendedAddressing(IREmitter *IREmit, uint8_t AccessSize, IROp_Header* AddressHeader) {
using MemExtendedAddrResult =
std::tuple<MemOffsetType, uint8_t, OrderedNode *, OrderedNode *>;
// If this optimization doesn't succeed, it will return the nullopt
static std::optional<MemExtendedAddrResult>
MemExtendedAddressing(IREmitter *IREmit, uint8_t AccessSize,
IROp_Header *AddressHeader) {
// Try to optimize: AddShift Base, LSHL(Offset, Scale)
if (AddressHeader->Op == OP_ADDSHIFT) {
auto AddShift = AddressHeader->C<IROp_AddShift>();
if (AddShift->Shift == IR::ShiftType::LSL) {
auto Scale = 1U << AddShift->ShiftAmount;
if (IsMemoryScale(Scale, AccessSize)) {
// remove shift as it can be folded to the mem op
return std::make_optional(
std::make_tuple(MEM_OFFSET_SXTX, (uint8_t)Scale,
IREmit->UnwrapNode(AddShift->Src2),
IREmit->UnwrapNode(AddShift->Src1)));
} else if (Scale == 1) {
return std::make_optional(std::make_tuple(
MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddShift->Src2),
IREmit->UnwrapNode(AddShift->Src1)));
}
}
return std::nullopt;
}
LOGMAN_THROW_A_FMT(AddressHeader->Op == OP_ADD, "Invalid address Op");
auto Src0Header = IREmit->GetOpHeader(AddressHeader->Args[0]);
if (Src0Header->Size == 8) {
//Try to optimize: Base + MUL(Offset, Scale)
// Try to optimize: Base + MUL(Offset, Scale)
if (Src0Header->Op == OP_MUL) {
uint64_t Scale;
if (IREmit->IsValueConstant(Src0Header->Args[1], &Scale)) {
if (IsMemoryScale(Scale, AccessSize)) {
// remove mul as it can be folded to the mem op
return { MEM_OFFSET_SXTX, (uint8_t)Scale, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
return std::make_optional(
std::make_tuple(MEM_OFFSET_SXTX, (uint8_t)Scale,
IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
} else if (Scale == 1) {
// remove nop mul
return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
return std::make_optional(std::make_tuple(
MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
}
}
}
//Try to optimize: Base + LSHL(Offset, Scale)
// Try to optimize: Base + LSHL(Offset, Scale)
else if (Src0Header->Op == OP_LSHL) {
uint64_t Constant2;
if (IREmit->IsValueConstant(Src0Header->Args[1], &Constant2)) {
uint64_t Scale = 1<<Constant2;
if (IsMemoryScale(Scale, AccessSize)) {
// remove shift as it can be folded to the mem op
return { MEM_OFFSET_SXTX, Scale, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
return std::make_optional(
std::make_tuple(MEM_OFFSET_SXTX, Scale,
IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
} else if (Scale == 1) {
// remove nop shift
return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
return std::make_optional(std::make_tuple(
MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
}
}
}
#if defined(_M_ARM_64) // x86 can't sext or zext on mem ops
//Try to optimize: Base + (u32)Offset
// Try to optimize: Base + (u32)Offset
else if (Src0Header->Op == OP_BFE) {
auto Bfe = Src0Header->C<IROp_Bfe>();
if (Bfe->lsb == 0 && Bfe->Width == 32) {
//todo: arm can also scale here
return { MEM_OFFSET_UXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
return std::make_optional(std::make_tuple(
MEM_OFFSET_UXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
}
}
//Try to optimize: Base + (s32)Offset
// Try to optimize: Base + (s32)Offset
else if (Src0Header->Op == OP_SBFE) {
auto Sbfe = Src0Header->C<IROp_Sbfe>();
if (Sbfe->lsb == 0 && Sbfe->Width == 32) {
//todo: arm can also scale here
return { MEM_OFFSET_SXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
// todo: arm can also scale here
return std::make_optional(std::make_tuple(
MEM_OFFSET_SXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
}
}
#endif
}
// no match anywhere, just add
return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[0]), IREmit->UnwrapNode(AddressHeader->Args[1]) };
// However, if we have one 32bit negative constant, we need to sign extend it
auto Arg0_ = AddressHeader->Args[0];
auto Arg1_ = AddressHeader->Args[1];
auto Arg1H = IREmit->GetOpHeader(Arg1_);
auto Arg0 = IREmit->UnwrapNode(Arg0_);
auto Arg1 = IREmit->UnwrapNode(Arg1_);
uint64_t ConstVal = 0;
// Only optimize in 32bits reg+const where const < 16Kb.
if (Arg1H->Size == 4 && IREmit->IsValueConstant(Arg1_, &ConstVal)) {
// Base is Arg0, Constant (Displacement in Arg1)
OrderedNode *Base = Arg0;
OrderedNode *Cnt = Arg1;
int32_t Val32 = (int32_t)ConstVal;
if (Val32 > -16384 && Val32 < 0) {
return std::make_optional(std::make_tuple(MEM_OFFSET_SXTW, 1, Base, Cnt));
} else if (Val32 >= 0 && Val32 < 16384) {
return std::make_optional(std::make_tuple(MEM_OFFSET_SXTX, 1, Base, Cnt));
}
} else if (AddressHeader->Size == 4) {
// Do not optimize 32bit reg+reg.
// Something like :
// add w20, w7, w5
// ldr w7, [x20]
//
// cannot be simplified to (or any other single load instruction)
// ldr w7, [x5, w7, sxtx]
return std::nullopt;
} else {
return std::make_optional(std::make_tuple(MEM_OFFSET_SXTX, 1, Arg0, Arg1));
}
return std::nullopt;
}
static OrderedNodeWrapper RemoveUselessMasking(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t mask) {
@@ -184,9 +258,10 @@ static bool IsBfeAlreadyDone(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t
class ConstProp final : public FEXCore::IR::Pass {
public:
explicit ConstProp(bool DoInlineConstants, bool SupportsTSOImm9)
: InlineConstants(DoInlineConstants)
, SupportsTSOImm9 {SupportsTSOImm9} { }
explicit ConstProp(bool DoInlineConstants, bool SupportsTSOImm9,
bool Is64BitMode)
: InlineConstants(DoInlineConstants), SupportsTSOImm9{SupportsTSOImm9},
Is64BitMode(Is64BitMode) {}
bool Run(IREmitter *IREmit) override;
@@ -219,6 +294,7 @@ private:
return Result.first->second;
}
bool SupportsTSOImm9{};
bool Is64BitMode;
// This is a heuristic to limit constant pool live ranges to reduce RA interference pressure.
// If the range is unbounded then RA interference pressure seems to increase to the point
// that long blocks of constant usage can slow to a crawl.
@@ -439,50 +515,6 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
bool Changed = false;
switch (IROp->Op) {
/*
case OP_UMUL:
case OP_DIV:
case OP_UDIV:
case OP_REM:
case OP_UREM:
case OP_MULH:
case OP_UMULH:
case OP_LSHR:
case OP_ASHR:
case OP_ROL:
case OP_ROR:
case OP_LDIV:
case OP_LUDIV:
case OP_LREM:
case OP_LUREM:
case OP_BFI:
{
uint64_t Constant1;
uint64_t Constant2;
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) &&
IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
LOGMAN_MSG_A_FMT("Could const prop op: {}", IR::GetName(IROp->Op));
}
break;
}
case OP_SEXT:
case OP_NEG:
case OP_POPCOUNT:
case OP_FINDLSB:
case OP_FINDMSB:
case OP_REV:
case OP_SBFE: {
uint64_t Constant1;
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
LOGMAN_MSG_A_FMT("Could const prop op: {}", IR::GetName(IROp->Op));
}
break;
}
*/
case OP_LOADMEMTSO: {
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
@@ -490,8 +522,12 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
// Support once hardware is available to use this.
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
@@ -509,8 +545,12 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
// Support once hardware is available to use this.
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
@@ -525,12 +565,19 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
auto Op = IROp->CW<IR::IROp_LoadMem>();
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (AddressHeader->Op == OP_ADD &&
((Is64BitMode && AddressHeader->Size == 8) ||
(!Is64BitMode && AddressHeader->Size == 4))) {
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
Changed = true;
@@ -542,8 +589,15 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
auto Op = IROp->CW<IR::IROp_StoreMem>();
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (AddressHeader->Op == OP_ADD &&
((Is64BitMode && AddressHeader->Size == 8) ||
(!Is64BitMode && AddressHeader->Size == 4))) {
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
@@ -555,23 +609,65 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
break;
}
case OP_ADD: {
case OP_PREFETCH: {
auto Op = IROp->CW<IR::IROp_Prefetch>();
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
const bool SupportedOp =
AddressHeader->Op == OP_ADD ||
AddressHeader->Op == OP_ADDSHIFT;
if (SupportedOp &&
((Is64BitMode && AddressHeader->Size == 8) ||
(!Is64BitMode && AddressHeader->Size == 4))) {
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
Changed = true;
}
break;
}
case OP_ADD:
case OP_SUB:
case OP_ADDWITHFLAGS:
case OP_SUBWITHFLAGS: {
auto Op = IROp->C<IR::IROp_Add>();
uint64_t Constant1{};
uint64_t Constant2{};
bool IsConstant1 = IREmit->IsValueConstant(Op->Header.Args[0], &Constant1);
bool IsConstant2 = IREmit->IsValueConstant(Op->Header.Args[1], &Constant2);
if (IsConstant1 && IsConstant2) {
if (IsConstant1 && IsConstant2 && IROp->Op == OP_ADD) {
uint64_t NewConstant = (Constant1 + Constant2) & getMask(Op) ;
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
Changed = true;
} else if (IsConstant1 && IsConstant2 && IROp->Op == OP_SUB) {
uint64_t NewConstant = (Constant1 - Constant2) & getMask(Op) ;
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
Changed = true;
}
else if (IsConstant2 && !IsImmAddSub(Constant2) && IsImmAddSub(-Constant2)) {
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
// This means we can convert the operation in to a subtract.
// Change the IR operation itself.
IROp->Op = OP_SUB;
// So, negate the operation to negate (and inline) the constant.
if (IROp->Op == OP_ADD)
IROp->Op = OP_SUB;
else if (IROp->Op == OP_SUB)
IROp->Op = OP_ADD;
else if (IROp->Op == OP_ADDWITHFLAGS)
IROp->Op = OP_SUBWITHFLAGS;
else if (IROp->Op == OP_SUBWITHFLAGS)
IROp->Op = OP_ADDWITHFLAGS;
// Set the write cursor to just before this operation.
auto CodeIter = CurrentIR.at(CodeNode);
--CodeIter;
@@ -586,19 +682,6 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
}
break;
}
case OP_SUB: {
auto Op = IROp->C<IR::IROp_Sub>();
uint64_t Constant1{};
uint64_t Constant2{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
uint64_t NewConstant = (Constant1 - Constant2) & getMask(Op) ;
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
Changed = true;
}
break;
}
case OP_SUBSHIFT: {
auto Op = IROp->C<IR::IROp_SubShift>();
@@ -645,23 +728,6 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
}
break;
}
/* TODO: restore this when we have rmif or something? */
#if 0
case OP_TESTNZ: {
auto Op = IROp->CW<IR::IROp_TestNZ>();
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
bool N = Constant1 & (1ull << ((Op->Size * 8) - 1));
bool Z = Constant1 == 0;
uint32_t NZVC = (N ? (1u << 31) : 0) | (Z ? (1u << 30) : 0);
IREmit->ReplaceWithConstant(CodeNode, NZVC);
Changed = true;
}
break;
}
#endif
case OP_OR: {
auto Op = IROp->CW<IR::IROp_Or>();
uint64_t Constant1{};
@@ -738,6 +804,17 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
}
break;
}
case OP_NEG: {
auto Op = IROp->CW<IR::IROp_Neg>();
uint64_t Constant{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant)) {
uint64_t NewConstant = -Constant;
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
Changed = true;
}
break;
}
case OP_LSHL: {
auto Op = IROp->CW<IR::IROp_Lshl>();
uint64_t Constant1{};
@@ -1011,7 +1088,23 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
break;
}
case OP_RMIFNZCV:
{
auto Op = IROp->C<IR::IROp_RmifNZCV>();
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
if (Constant1 == 0) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
Changed = true;
}
}
break;
}
case OP_CONDADDNZCV:
case OP_CONDSUBNZCV:
{
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
@@ -1068,17 +1161,12 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
}
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
#ifdef JIT_ARM64
bool SupportsAllOnes = true;
#else
bool SupportsAllOnes = false;
#endif
uint64_t Constant2{};
uint64_t Constant3{};
if (IREmit->IsValueConstant(Op->Header.Args[2], &Constant2) &&
IREmit->IsValueConstant(Op->Header.Args[3], &Constant3) &&
(Constant2 == 1 || (SupportsAllOnes && Constant2 == AllOnes)) &&
(Constant2 == 1 || Constant2 == AllOnes) &&
Constant3 == 0)
{
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[2]));
@@ -1252,7 +1340,7 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant));
Changed = true;
}
@@ -1266,13 +1354,29 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant));
Changed = true;
}
break;
}
case OP_PREFETCH:
{
auto Op = IROp->CW<IR::IROp_Prefetch>();
uint64_t Constant2{};
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) {
if (IsImmMemory(Constant2, IROp->Size)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset));
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, CreateInlineConstant(IREmit, Constant2));
Changed = true;
}
}
break;
}
default:
break;
}
@@ -1311,8 +1415,9 @@ bool ConstProp::Run(IREmitter *IREmit) {
return Changed;
}
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants, bool SupportsTSOImm9) {
return fextl::make_unique<ConstProp>(InlineConstants, SupportsTSOImm9);
fextl::unique_ptr<FEXCore::IR::Pass>
CreateConstProp(bool InlineConstants, bool SupportsTSOImm9, bool Is64BitMode) {
return fextl::make_unique<ConstProp>(InlineConstants, SupportsTSOImm9,
Is64BitMode);
}
}
@@ -9,7 +9,6 @@ $end_info$
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
@@ -6,13 +6,14 @@ desc: Transforms ContextLoad/Store to temporaries, similar to mem2reg
$end_info$
*/
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/EnumOperators.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
@@ -483,6 +484,8 @@ private:
ContextMemberInfo *RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode = nullptr);
ContextMemberInfo *RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode = nullptr);
bool HandleLoadFlag(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::OrderedNode *CodeNode, unsigned Flag);
// Classify context loads and stores.
bool ClassifyContextLoad(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::NodeIterator BlockEnd);
bool ClassifyContextStore(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::OrderedNode *ValueNode);
@@ -544,10 +547,42 @@ bool RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter *IREmit, ContextInfo *Loc
bool RCLSE::ClassifyContextStore(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::OrderedNode *ValueNode) {
auto Info = FindMemberInfo(LocalInfo, Offset, Size);
ContextMemberInfo PreviousMemberInfoCopy = *Info;
RecordAccess(Info, Class, Offset, Size, LastAccessType::WRITE, ValueNode,
CodeNode);
// TODO: Optimize redundant stores.
// ContextMemberInfo PreviousMemberInfoCopy = *Info;
if (PreviousMemberInfoCopy.AccessRegClass == Info->AccessRegClass &&
PreviousMemberInfoCopy.AccessOffset == Info->AccessOffset &&
PreviousMemberInfoCopy.AccessSize == Size &&
PreviousMemberInfoCopy.Accessed == LastAccessType::WRITE) {
// This optimizes redundant stores with no intervening load
IREmit->Remove(PreviousMemberInfoCopy.StoreNode);
return true;
}
// TODO: Optimize the case of partial stores.
return false;
}
bool RCLSE::HandleLoadFlag(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::OrderedNode *CodeNode, unsigned Flag) {
const auto FlagOffset = offsetof(FEXCore::Core::CPUState, flags[Flag]);
auto Info = FindMemberInfo(LocalInfo, FlagOffset, 1);
LastAccessType LastAccess = Info->Accessed;
auto LastValueNode = Info->ValueNode;
if (IsWriteAccess(LastAccess)) { // 1 byte so always a full write
// If the last store matches this load value then we can replace the loaded value with the previous valid one
IREmit->SetWriteCursor(CodeNode);
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
return true;
}
else if (IsReadAccess(LastAccess)) {
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
return true;
}
return false;
}
@@ -660,23 +695,11 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
}
else if (IROp->Op == OP_LOADFLAG) {
const auto Op = IROp->CW<IR::IROp_LoadFlag>();
const auto FlagOffset = offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag;
auto Info = FindMemberInfo(&LocalInfo, FlagOffset, 1);
LastAccessType LastAccess = Info->Accessed;
OrderedNode *LastValueNode = Info->ValueNode;
if (IsWriteAccess(LastAccess)) { // 1 byte so always a full write
// If the last store matches this load value then we can replace the loaded value with the previous valid one
IREmit->SetWriteCursor(CodeNode);
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
Changed = true;
}
else if (IsReadAccess(LastAccess)) {
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
Changed = true;
}
Changed |= HandleLoadFlag(IREmit, &LocalInfo, CodeNode, Op->Flag);
}
else if (IROp->Op == OP_LOADDF) {
Changed |= HandleLoadFlag(IREmit, &LocalInfo, CodeNode, X86State::RFLAG_DF_RAW_LOC);
}
else if (IROp->Op == OP_SYSCALL ||
IROp->Op == OP_INLINESYSCALL) {
@@ -10,8 +10,8 @@ $end_info$
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/unordered_map.h>
@@ -211,6 +211,10 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
auto& BlockInfo = InfoMap[BlockNode];
BlockInfo.flag.reads |= 1UL << Op->Flag;
} else if (IROp->Op == OP_LOADDF) {
auto& BlockInfo = InfoMap[BlockNode];
BlockInfo.flag.reads |= 1UL << X86State::RFLAG_DF_RAW_LOC;
} else if (IROp->Op == OP_STOREREGISTER) {
auto Op = IROp->C<IR::IROp_StoreRegister>();
@@ -11,7 +11,6 @@ $end_info$
#include "Interface/Core/OpcodeDispatcher.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/Profiler.h>
@@ -9,12 +9,11 @@ $end_info$
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Interface/IR/Passes/IRValidation.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/sstream.h>
@@ -11,7 +11,6 @@ $end_info$
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Profiler.h>
@@ -9,7 +9,6 @@ $end_info$
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
@@ -3,12 +3,11 @@
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Interface/IR/Passes/IRValidation.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/deque.h>
#include <FEXCore/fextl/fmt.h>
@@ -10,7 +10,6 @@ $end_info$
#include "Interface/IR/IREmitter.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/Profiler.h>
#include "Interface/IR/PassManager.h"
@@ -176,6 +175,12 @@ DeadFlagCalculationEliminination::Classify(IROp_Header *IROp)
.CanEliminate = true,
};
case OP_CMPPAIRZ:
return {
.Write = FLAG_Z,
.CanEliminate = true,
};
case OP_CARRYINVERT:
return {
.Read = FLAG_C,
@@ -222,6 +227,7 @@ DeadFlagCalculationEliminination::Classify(IROp_Header *IROp)
return {.Read = FlagsForCondClassType(Op->Cond)};
}
case OP_CONDSUBNZCV:
case OP_CONDADDNZCV: {
auto Op = IROp->CW<IR::IROp_CondAddNZCV>();
return {
@@ -9,15 +9,16 @@ $end_info$
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "FEXCore/Core/X86Enums.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Interface/IR/Passes.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/unordered_map.h>
@@ -25,7 +26,6 @@ $end_info$
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <algorithm>
#include <cstddef>
@@ -68,7 +68,7 @@ namespace {
};
static_assert(sizeof(RegisterNode) == 128 * 4);
constexpr size_t REGISTER_NODES_PER_PAGE = FHU::FEX_PAGE_SIZE / sizeof(RegisterNode);
constexpr size_t REGISTER_NODES_PER_PAGE = FEXCore::Utils::FEX_PAGE_SIZE / sizeof(RegisterNode);
struct RegisterSet {
fextl::vector<RegisterClass> Classes;
@@ -240,8 +240,6 @@ namespace {
RegisterAllocationData::UniquePtr PullAllocationData() override;
private:
using BlockInterferences = fextl::vector<IR::NodeID>;
IR::NodeID SpillPointId;
fextl::vector<BucketList<DEFAULT_INTERFERENCE_SPAN_COUNT, uint32_t>> SpanStart;
@@ -253,9 +251,6 @@ namespace {
fextl::vector<LiveRange> LiveRanges;
fextl::unordered_map<IR::NodeID, BlockInterferences> LocalBlockInterferences;
BlockInterferences GlobalBlockInterferences;
[[nodiscard]] static constexpr uint32_t InfoMake(uint32_t id, uint32_t Class) {
return id | (Class << 24);
}
@@ -273,8 +268,6 @@ namespace {
void CalculateLiveRange(FEXCore::IR::IRListView *IR);
void OptimizeStaticRegisters(FEXCore::IR::IRListView *IR);
void CalculateBlockInterferences(FEXCore::IR::IRListView *IR);
void CalculateBlockNodeInterference(FEXCore::IR::IRListView *IR);
void CalculateNodeInterference(FEXCore::IR::IRListView *IR);
void AllocateVirtualRegisters();
void CalculatePredecessors(FEXCore::IR::IRListView *IR);
@@ -295,6 +288,10 @@ namespace {
uint32_t FindSpillSlot(IR::NodeID Node, FEXCore::IR::RegisterClassType RegisterClass);
bool RunAllocateVirtualRegisters(IREmitter *IREmit);
uint64_t OriginalRIP;
fextl::vector<LiveRange*> StaticMaps;
};
ConstrainedRAPass::ConstrainedRAPass(FEXCore::IR::Pass* _CompactionPass, bool _SupportsAVX)
@@ -548,7 +545,7 @@ namespace {
auto GprSize = Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount;
auto MapsSize = Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount + Graph->Set.Classes[FPRFixedClass.Val].PhysicalCount;
LiveRange* StaticMaps[MapsSize];
StaticMaps.resize(MapsSize);
// Get a StaticMap entry from context offset
const auto GetStaticMapFromOffset = [&](uint32_t Offset) -> LiveRange** {
@@ -623,7 +620,7 @@ namespace {
// - Mark read-aliases
// - Demote read-aliases if SRA reg is written before the alias's last read
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
memset(StaticMaps, 0, MapsSize * sizeof(LiveRange*));
memset(StaticMaps.data(), 0, MapsSize * sizeof(LiveRange*));
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
const auto Node = IR->GetID(CodeNode);
auto& NodeLiveRange = LiveRanges[Node.Value];
@@ -745,103 +742,6 @@ namespace {
}
}
void ConstrainedRAPass::CalculateBlockInterferences(FEXCore::IR::IRListView *IR) {
using namespace FEXCore;
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
const auto BlockNodeID = IR->GetID(BlockNode);
const auto BlockBeginID = BlockIROp->Begin.ID();
const auto BlockLastID = BlockIROp->Last.ID();
auto& BlockInterferenceVector = LocalBlockInterferences.try_emplace(BlockNodeID).first->second;
BlockInterferenceVector.reserve(BlockLastID.Value - BlockBeginID.Value);
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
const auto Node = IR->GetID(CodeNode);
LiveRange& NodeLiveRange = LiveRanges[Node.Value];
if (NodeLiveRange.Begin >= BlockBeginID &&
NodeLiveRange.End <= BlockLastID) {
// If the live range of this node is FULLY inside of the block
// Then add it to the block specific interference list
BlockInterferenceVector.emplace_back(Node);
}
else {
// If the live range is not fully inside the block then add it to the global interference list
GlobalBlockInterferences.emplace_back(Node);
}
}
}
}
void ConstrainedRAPass::CalculateBlockNodeInterference(FEXCore::IR::IRListView *IR) {
#if 0
const auto AddInterference = [&](IR::NodeID Node1, IR::NodeID Node2) {
RegisterNode *Node = &Graph->Nodes[Node1.Value];
Node->Interference.Set(Node2);
Node->InterferenceList[Node->Head.InterferenceCount++] = Node2;
};
const auto CheckInterferenceNodeSizes = [&](IR::NodeID Node1, uint32_t MaxNewNodes) {
RegisterNode *Node = &Graph->Nodes[Node1.Value];
uint32_t NewListMax = Node->Head.InterferenceCount + MaxNewNodes;
if (Node->InterferenceListSize <= NewListMax) {
const auto AlignedListCount = static_cast<uint32_t>(FEXCore::AlignUp(NewListMax, DEFAULT_INTERFERENCE_LIST_COUNT));
Node->InterferenceListSize = std::max(Node->InterferenceListSize * 2U, AlignedListCount);
Node->InterferenceList = reinterpret_cast<uint32_t*>(realloc(Node->InterferenceList, Node->InterferenceListSize * sizeof(uint32_t)));
}
};
using namespace FEXCore;
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
BlockInterferences *BlockInterferenceVector = &LocalBlockInterferences.try_emplace(IR->GetID(BlockNode)).first->second;
fextl::vector<IR::NodeID> Interferences;
Interferences.reserve(BlockInterferenceVector->size() + GlobalBlockInterferences.size());
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
const auto Node = IR->GetID(CodeNode);
const auto& NodeLiveRange = LiveRanges[Node.Value];
// Check for every interference with the local block's interference
for (auto RHSNode : *BlockInterferenceVector) {
const auto& RHSNodeLiveRange = LiveRanges[RHSNode.Value];
if (!(NodeLiveRange.Begin >= RHSNodeLiveRange.End ||
RHSNodeLiveRange.Begin >= NodeLiveRange.End)) {
Interferences.emplace_back(RHSNode);
}
}
// Now check the global block interference vector
for (auto RHSNode : GlobalBlockInterferences) {
const auto& RHSNodeLiveRange = LiveRanges[RHSNode.Value];
if (!(NodeLiveRange.Begin >= RHSNodeLiveRange.End ||
RHSNodeLiveRange.Begin >= NodeLiveRange.End)) {
Interferences.emplace_back(RHSNode);
}
}
CheckInterferenceNodeSizes(Node, Interferences.size());
for (auto RHSNode : Interferences) {
AddInterference(Node, RHSNode);
}
for (auto RHSNode : Interferences) {
AddInterference(RHSNode, Node);
CheckInterferenceNodeSizes(RHSNode, 0);
}
Interferences.clear();
}
}
#endif
}
void ConstrainedRAPass::CalculateNodeInterference(FEXCore::IR::IRListView *IR) {
const auto AddInterference = [this](IR::NodeID Node1, IR::NodeID Node2) {
RegisterNode *Node = &Graph->Nodes[Node1.Value];
@@ -1227,7 +1127,7 @@ namespace {
if (!CurrentNodes.contains(InterferenceNode)) {
InterferenceIdToSpill = InterferenceNode;
LogMan::Msg::DFmt("Panic spilling %{}, Live Range[{}, {})", InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
LogMan::Msg::DFmt("[RIP: 0x{:x}] Panic spilling %{}, Live Range[{}, {})", OriginalRIP, InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
return true;
}
return false;
@@ -1395,9 +1295,6 @@ namespace {
using namespace FEXCore;
bool Changed = false;
GlobalBlockInterferences.clear();
LocalBlockInterferences.clear();
auto IR = IREmit->ViewIR();
uint32_t SSACount = IR.GetSSACount();
@@ -1406,16 +1303,7 @@ namespace {
FindNodeClasses(Graph, &IR);
CalculateLiveRange(&IR);
OptimizeStaticRegisters(&IR);
// Linear forward scan based interference calculation is faster for smaller blocks
// Smarter block based interference calculation is faster for larger blocks
/*if (SSACount >= 2048) {
CalculateBlockInterferences(&IR);
CalculateBlockNodeInterference(&IR);
}
else*/ {
CalculateNodeInterference(&IR);
}
CalculateNodeInterference(&IR);
AllocateVirtualRegisters();
return Changed;
@@ -1446,6 +1334,8 @@ namespace {
auto IR = IREmit->ViewIR();
auto HeaderOp = IR.GetHeader();
OriginalRIP = HeaderOp->OriginalRIP;
SpillSlotCount = 0;
Graph->SpillStack.clear();
@@ -11,7 +11,6 @@ $end_info$
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/set.h>
+103 -112
View File
@@ -3,13 +3,14 @@
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/memory_resource.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <array>
#include <cctype>
#include <charconv>
#include <cstdio>
#include <fcntl.h>
#ifndef _WIN32
@@ -94,8 +95,8 @@ namespace FEXCore::Allocator {
// Now allocate the next page after the sbrk address to ensure it can't grow.
// In most cases at the start of `main` this will already be page aligned, which means subsequent `sbrk`
// calls won't allocate any memory through that.
void* AlignedBRK = reinterpret_cast<void*>(FEXCore::AlignUp(reinterpret_cast<uintptr_t>(StartingSBRK), FHU::FEX_PAGE_SIZE));
void *AfterBRK = mmap(AlignedBRK, FHU::FEX_PAGE_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED_NOREPLACE | MAP_NORESERVE, -1, 0);
void* AlignedBRK = reinterpret_cast<void*>(FEXCore::AlignUp(reinterpret_cast<uintptr_t>(StartingSBRK), FEXCore::Utils::FEX_PAGE_SIZE));
void *AfterBRK = mmap(AlignedBRK, FEXCore::Utils::FEX_PAGE_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED_NOREPLACE | MAP_NORESERVE, -1, 0);
if (AfterBRK == INVALID_PTR) {
// Couldn't allocate the page after the aligned brk? This should never happen.
// FEXCore::LogMan isn't configured yet so we just need to print the message.
@@ -117,7 +118,7 @@ namespace FEXCore::Allocator {
void ReenableSBRKAllocations(void* Ptr) {
const void* INVALID_PTR = reinterpret_cast<void*>(~0ULL);
if (Ptr != INVALID_PTR) {
munmap(Ptr, FHU::FEX_PAGE_SIZE);
munmap(Ptr, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
@@ -171,10 +172,10 @@ namespace FEXCore::Allocator {
for (int i = 0; i < 64; ++i) {
// Try grabbing a some of the top pages of the range
// x86 allocates some high pages in the top end
void *Ptr = ::mmap(reinterpret_cast<void*>(Size - FHU::FEX_PAGE_SIZE * i), FHU::FEX_PAGE_SIZE, PROT_NONE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
void *Ptr = ::mmap(reinterpret_cast<void*>(Size - FEXCore::Utils::FEX_PAGE_SIZE * i), FEXCore::Utils::FEX_PAGE_SIZE, PROT_NONE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
if (Ptr != (void*)~0ULL) {
::munmap(Ptr, FHU::FEX_PAGE_SIZE);
if (Ptr == (void*)(Size - FHU::FEX_PAGE_SIZE * i)) {
::munmap(Ptr, FEXCore::Utils::FEX_PAGE_SIZE);
if (Ptr == (void*)(Size - FEXCore::Utils::FEX_PAGE_SIZE * i)) {
return true;
}
}
@@ -194,141 +195,131 @@ namespace FEXCore::Allocator {
#define STEAL_LOG(...) // fprintf(stderr, __VA_ARGS__)
fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
void * const StackLocation = alloca(0);
const uintptr_t StackLocation_u64 = reinterpret_cast<uintptr_t>(StackLocation);
fextl::vector<MemoryRegion> CollectMemoryGaps(uintptr_t Begin, uintptr_t End, int MapsFD) {
fextl::vector<MemoryRegion> Regions;
int MapsFD = open("/proc/self/maps", O_RDONLY);
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
enum {ParseBegin, ParseEnd, ScanEnd} State = ParseBegin;
uintptr_t RegionBegin = 0;
uintptr_t RegionEnd = 0;
uintptr_t PreviousMapEnd = 0;
char Buffer[2048];
const char *Cursor;
const char *Cursor = Buffer;
ssize_t Remaining = 0;
for(;;) {
bool EndOfFileReached = false;
if (Remaining == 0) {
while (true) {
const auto line_begin = Cursor;
auto line_end = std::find(line_begin, Cursor + Remaining, '\n');
// Check if the buffered data covers the entire line.
// If not, try buffering more data.
if (line_end == Cursor + Remaining) {
if (EndOfFileReached) {
// No more data to buffer. Add remaining memory and return.
const auto MapBegin = std::max(RegionEnd, Begin);
STEAL_LOG("[%d] EndOfFile; MapBegin: %016lX MapEnd: %016lX\n", __LINE__, MapBegin, End);
if (End > MapBegin) {
Regions.push_back({(void*)MapBegin, End - MapBegin});
}
return Regions;
}
// Move pending content back to the beginning, then buffer more data.
std::copy(Cursor, Cursor + Remaining, std::begin(Buffer));
auto PendingBytes = Remaining;
do {
Remaining = read(MapsFD, Buffer, sizeof(Buffer));
} while ( Remaining == -1 && errno == EAGAIN);
Remaining = read(MapsFD, Buffer + PendingBytes, sizeof(Buffer) - PendingBytes);
} while (Remaining == -1 && errno == EAGAIN);
if (Remaining < sizeof(Buffer) - PendingBytes) {
EndOfFileReached = true;
}
Remaining += PendingBytes;
Cursor = Buffer;
}
if (Remaining == 0 && State == ParseBegin) {
STEAL_LOG("[%d] EndOfFile; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
const auto MapBegin = std::max(RegionEnd, Begin);
const auto MapEnd = End;
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
if (MapEnd > MapBegin) {
STEAL_LOG(" Reserving\n");
auto MapSize = MapEnd - MapBegin;
auto Alloc = mmap((void*)MapBegin, MapSize, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", MapBegin, MapSize);
LogMan::Throw::AFmt(Alloc == (void*)MapBegin, "mmap({},{:x}) returned {} instead of {:x}", Alloc, MapBegin);
Regions.push_back({(void*)MapBegin, MapSize});
}
close(MapsFD);
return Regions;
}
LogMan::Throw::AFmt(Remaining > 0, "Failed to parse /proc/self/maps");
auto c = *Cursor++;
Remaining--;
if (State == ScanEnd) {
if (c == '\n') {
State = ParseBegin;
}
continue;
}
if (State == ParseBegin) {
if (c == '-') {
STEAL_LOG("[%d] ParseBegin; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
// Parse mapped region in the format "fffff7cc3000-fffff7cc4000 r--p ..."
{
uintptr_t RegionBegin;
auto result = std::from_chars(Cursor, line_end, RegionBegin, 16);
LogMan::Throw::AFmt(result.ec == std::errc{} && *result.ptr == '-', "Unexpected line format");
Cursor = result.ptr + 1;
const auto MapBegin = std::max(RegionEnd, Begin);
const auto MapEnd = std::min(RegionBegin, End);
// Add gap between the previous region and the current one
const auto MapBegin = std::max(RegionEnd, Begin);
const auto MapEnd = std::min(RegionBegin, End);
if (MapEnd > MapBegin) {
Regions.push_back({(void*)MapBegin, MapEnd - MapBegin});
}
// Store the location we are going to map.
PreviousMapEnd = MapEnd;
result = std::from_chars(Cursor, line_end, RegionEnd, 16);
LogMan::Throw::AFmt(result.ec == std::errc{} && *result.ptr == ' ', "Unexpected line format");
Cursor = result.ptr + 1;
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
STEAL_LOG("[%d] parsed line: RegionBegin=%016lX RegionEnd=%016lX\n", __LINE__, RegionBegin, RegionEnd);
if (MapEnd > MapBegin) {
STEAL_LOG(" Reserving\n");
auto MapSize = MapEnd - MapBegin;
auto Alloc = mmap((void*)MapBegin, MapSize, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", MapBegin, MapSize);
LogMan::Throw::AFmt(Alloc == (void*)MapBegin, "mmap({},{:x}) returned {} instead of {:x}", Alloc, MapBegin);
Regions.push_back({(void*)MapBegin, MapSize});
}
RegionBegin = 0;
RegionEnd = 0;
State = ParseEnd;
continue;
} else {
LogMan::Throw::AFmt(std::isalpha(c) || std::isdigit(c), "Unexpected char '{}' in ParseBegin", c);
RegionBegin = (RegionBegin << 4) | (c <= '9' ? (c - '0') : (c - 'a' + 10));
if (RegionEnd >= End) {
// Early return if we are completely beyond the allocation space.
return Regions;
}
}
if (State == ParseEnd) {
if (c == ' ') {
STEAL_LOG("[%d] ParseEnd; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
Remaining -= line_end + 1 - line_begin;
Cursor = line_end + 1;
}
FEX_UNREACHABLE;
}
if (RegionEnd > End) {
// Early return if we are completely beyond the allocation space.
close(MapsFD);
return Regions;
}
fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
const uintptr_t StackLocation_u64 = reinterpret_cast<uintptr_t>(alloca(0));
State = ScanEnd;
const int MapsFD = open("/proc/self/maps", O_RDONLY);
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
// If the previous map's ending and the region we just parsed overlap the stack then we need to save the stack mapping.
// Otherwise we will have severely limited stack size which crashes quickly.
if (PreviousMapEnd <= StackLocation_u64 && RegionEnd > StackLocation_u64) {
auto BelowStackRegion = Regions.back();
LOGMAN_THROW_AA_FMT(reinterpret_cast<uint64_t>(BelowStackRegion.Ptr) + BelowStackRegion.Size == PreviousMapEnd,
"This needs to match");
auto Regions = CollectMemoryGaps(Begin, End, MapsFD);
close(MapsFD);
// Allocate the region under the stack as READ | WRITE so the stack can still grow
auto Alloc = mmap(BelowStackRegion.Ptr, BelowStackRegion.Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
// If the memory bounds include the stack, blocking all memory regions will
// limit the stack size to the current value. To allow some stack growth,
// we don't block the memory gap directly below the stack memory but
// instead map it as readable+writable.
{
auto StackRegionIt =
std::find_if(Regions.begin(), Regions.end(),
[StackLocation_u64](auto& Region) {
return reinterpret_cast<uintptr_t>(Region.Ptr) + Region.Size > StackLocation_u64;
});
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", BelowStackRegion.Ptr, BelowStackRegion.Size);
LogMan::Throw::AFmt(Alloc == BelowStackRegion.Ptr, "mmap({},{:x}) returned {} instead of {:x}", Alloc, BelowStackRegion.Ptr);
// If no gap crossing the stack pointer was found but the SP is within
// the given bounds, the stack mapping is right after the last gap.
bool IsStackMapping = StackRegionIt != Regions.end() || StackLocation_u64 <= End;
Regions.pop_back();
}
continue;
} else {
LogMan::Throw::AFmt(std::isalpha(c) || std::isdigit(c), "Unexpected char '{}' in ParseEnd", c);
RegionEnd = (RegionEnd << 4) | (c <= '9' ? (c - '0') : (c - 'a' + 10));
}
if (IsStackMapping && StackRegionIt != Regions.begin() &&
reinterpret_cast<uintptr_t>(std::prev(StackRegionIt)->Ptr) + std::prev(StackRegionIt)->Size <= End) {
// Allocate the region under the stack as READ | WRITE so the stack can still grow
--StackRegionIt;
auto Alloc = mmap(StackRegionIt->Ptr, StackRegionIt->Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", StackRegionIt->Ptr, StackRegionIt->Size);
LogMan::Throw::AFmt(Alloc == StackRegionIt->Ptr, "mmap returned {} instead of {}", Alloc, fmt::ptr(StackRegionIt->Ptr));
Regions.erase(StackRegionIt);
}
}
ERROR_AND_DIE_FMT("unreachable");
// Block remaining memory gaps
for (auto RegionIt = Regions.begin(); RegionIt != Regions.end(); ++RegionIt) {
auto Alloc = mmap(RegionIt->Ptr, RegionIt->Size, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", RegionIt->Ptr, RegionIt->Size);
LogMan::Throw::AFmt(Alloc == RegionIt->Ptr, "mmap returned {} instead of {}", Alloc, fmt::ptr(RegionIt->Ptr));
}
return Regions;
}
fextl::vector<MemoryRegion> Steal48BitVA() {
@@ -6,9 +6,9 @@
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/vector.h>
@@ -70,8 +70,8 @@ namespace Alloc::OSAllocator {
// Lower bound is the starting of the range just past the lower 32bits
constexpr static uintptr_t LOWER_BOUND = 0x1'0000'0000ULL;
uintptr_t UPPER_BOUND_PAGE = UPPER_BOUND / FHU::FEX_PAGE_SIZE;
constexpr static uintptr_t LOWER_BOUND_PAGE = LOWER_BOUND / FHU::FEX_PAGE_SIZE;
uintptr_t UPPER_BOUND_PAGE = UPPER_BOUND / FEXCore::Utils::FEX_PAGE_SIZE;
constexpr static uintptr_t LOWER_BOUND_PAGE = LOWER_BOUND / FEXCore::Utils::FEX_PAGE_SIZE;
struct ReservedVMARegion {
uintptr_t Base;
@@ -114,22 +114,22 @@ namespace Alloc::OSAllocator {
// 0x100'0000 Pages
// 1 bit per page for tracking means 0x20'0000 (Pages / 8) bytes of flex space
// Which is 2MB of tracking
uint64_t NumElements = (Size >> FHU::FEX_PAGE_SHIFT) * sizeof(FlexBitElementType);
uint64_t NumElements = (Size >> FEXCore::Utils::FEX_PAGE_SHIFT) * sizeof(FlexBitElementType);
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<FlexBitElementType>::Size(NumElements);
}
static void InitializeVMARegionUsed(LiveVMARegion *Region, size_t AdditionalSize) {
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(Region->SlabInfo->RegionSize), FHU::FEX_PAGE_SIZE);
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(Region->SlabInfo->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
size_t SizePlusManagedData = SizeOfLiveRegion + AdditionalSize;
Region->FreeSpace = Region->SlabInfo->RegionSize - SizePlusManagedData;
size_t NumManagedPages = SizePlusManagedData >> FHU::FEX_PAGE_SHIFT;
size_t ManagedSize = NumManagedPages << FHU::FEX_PAGE_SHIFT;
size_t NumManagedPages = SizePlusManagedData >> FEXCore::Utils::FEX_PAGE_SHIFT;
size_t ManagedSize = NumManagedPages << FEXCore::Utils::FEX_PAGE_SHIFT;
// Use madvise to set the full tracking region to zero.
// This ensures unused pages are zero, while not having the backing pages consuming memory.
::madvise(Region->UsedPages.Memory + ManagedSize, (Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT) - ManagedSize, MADV_DONTNEED);
::madvise(Region->UsedPages.Memory + ManagedSize, (Region->SlabInfo->RegionSize >> FEXCore::Utils::FEX_PAGE_SHIFT) - ManagedSize, MADV_DONTNEED);
// Use madvise to claim WILLNEED on the beginning pages for initial state tracking.
// Improves performance of the following MemClear by not doing a page level fault dance for data necessary to track >170TB of used pages.
@@ -162,7 +162,7 @@ namespace Alloc::OSAllocator {
ReservedRegions->erase(ReservedIterator);
// mprotect the new region we've allocated
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(ReservedRegion->RegionSize), FHU::FEX_PAGE_SIZE);
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(ReservedRegion->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
[[maybe_unused]] auto Res = mprotect(reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData, PROT_READ | PROT_WRITE);
@@ -198,10 +198,10 @@ void OSAllocator_64Bit::DetermineVASize() {
UPPER_BOUND = Size;
#if _M_X86_64 // Last page cannot be allocated on x86
UPPER_BOUND -= FHU::FEX_PAGE_SIZE;
UPPER_BOUND -= FEXCore::Utils::FEX_PAGE_SIZE;
#endif
UPPER_BOUND_PAGE = UPPER_BOUND / FHU::FEX_PAGE_SIZE;
UPPER_BOUND_PAGE = UPPER_BOUND / FEXCore::Utils::FEX_PAGE_SIZE;
}
OSAllocator_64Bit::LiveVMARegion *OSAllocator_64Bit::FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd) {
@@ -250,13 +250,13 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
uint64_t Addr = reinterpret_cast<uint64_t>(addr);
// Addr must be page aligned
if (Addr & ~FHU::FEX_PAGE_MASK) {
if (Addr & ~FEXCore::Utils::FEX_PAGE_MASK) {
return reinterpret_cast<void*>(-EINVAL);
}
// If FD is provided then offset must also be page aligned
if (fd != -1 &&
offset & ~FHU::FEX_PAGE_MASK) {
offset & ~FEXCore::Utils::FEX_PAGE_MASK) {
return reinterpret_cast<void*>(-EINVAL);
}
@@ -266,10 +266,10 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
}
bool Fixed = (flags & MAP_FIXED) || (flags & MAP_FIXED_NOREPLACE);
length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE);
length = FEXCore::AlignUp(length, FEXCore::Utils::FEX_PAGE_SIZE);
uint64_t AddrEnd = Addr + length;
size_t NumberOfPages = length / FHU::FEX_PAGE_SIZE;
size_t NumberOfPages = length / FEXCore::Utils::FEX_PAGE_SIZE;
// This needs a mutex to be thread safe
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
@@ -285,14 +285,14 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
auto CheckIfRangeFits = [&AllocatedOffset](LiveVMARegion *Region, uint64_t length, int prot, int flags, int fd, off_t offset, uint64_t StartingPosition = 0) -> std::pair<LiveVMARegion*, void*> {
uint64_t AllocatedPage{~0ULL};
uint64_t NumberOfPages = length >> FHU::FEX_PAGE_SHIFT;
uint64_t NumberOfPages = length >> FEXCore::Utils::FEX_PAGE_SHIFT;
if (Region->FreeSpace >= length) {
uint64_t LastAllocation =
StartingPosition ?
(StartingPosition - Region->SlabInfo->Base) >> FHU::FEX_PAGE_SHIFT
(StartingPosition - Region->SlabInfo->Base) >> FEXCore::Utils::FEX_PAGE_SHIFT
: Region->LastPageAllocation;
size_t RegionNumberOfPages = Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT;
size_t RegionNumberOfPages = Region->SlabInfo->RegionSize >> FEXCore::Utils::FEX_PAGE_SHIFT;
if (Region->HadMunmap) {
@@ -317,7 +317,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
}
if (AllocatedPage != ~0ULL) {
AllocatedOffset = Region->SlabInfo->Base + AllocatedPage * FHU::FEX_PAGE_SIZE;
AllocatedOffset = Region->SlabInfo->Base + AllocatedPage * FEXCore::Utils::FEX_PAGE_SIZE;
// We need to setup protections for this
void *MMapResult = ::mmap(reinterpret_cast<void*>(AllocatedOffset),
@@ -407,7 +407,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
if (!LiveRegion) {
// Couldn't find a fit in the live regions
// Allocate a new reserved region
size_t lengthOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(length), FHU::FEX_PAGE_SIZE);
size_t lengthOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(length), FEXCore::Utils::FEX_PAGE_SIZE);
size_t lengthPlusManagedData = length + lengthOfLiveRegion;
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
if ((*it)->RegionSize >= lengthPlusManagedData) {
@@ -421,7 +421,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
if (LiveRegion) {
// Mark the pages as used
uintptr_t RegionBegin = LiveRegion->SlabInfo->Base;
uintptr_t MappedBegin = (AllocatedOffset - RegionBegin) >> FHU::FEX_PAGE_SHIFT;
uintptr_t MappedBegin = (AllocatedOffset - RegionBegin) >> FEXCore::Utils::FEX_PAGE_SHIFT;
for (size_t i = 0; i < NumberOfPages; ++i) {
LiveRegion->UsedPages.Set(MappedBegin + i);
@@ -447,11 +447,11 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
uint64_t Addr = reinterpret_cast<uint64_t>(addr);
if (Addr & ~FHU::FEX_PAGE_MASK) {
if (Addr & ~FEXCore::Utils::FEX_PAGE_MASK) {
return -EINVAL;
}
if (length & ~FHU::FEX_PAGE_MASK) {
if (length & ~FEXCore::Utils::FEX_PAGE_MASK) {
return -EINVAL;
}
@@ -462,7 +462,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
// This needs a mutex to be thread safe
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE);
length = FEXCore::AlignUp(length, FEXCore::Utils::FEX_PAGE_SIZE);
uintptr_t PtrBegin = reinterpret_cast<uintptr_t>(addr);
uintptr_t PtrEnd = PtrBegin + length;
@@ -476,8 +476,8 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
// Live region fully encompasses slab range
uint64_t FreedPages{};
uint32_t SlabPageBegin = (PtrBegin - RegionBegin) >> FHU::FEX_PAGE_SHIFT;
uint64_t PagesToFree = length >> FHU::FEX_PAGE_SHIFT;
uint32_t SlabPageBegin = (PtrBegin - RegionBegin) >> FEXCore::Utils::FEX_PAGE_SHIFT;
uint64_t PagesToFree = length >> FEXCore::Utils::FEX_PAGE_SHIFT;
for (size_t i = 0; i < PagesToFree; ++i) {
FreedPages += (*it)->UsedPages.TestAndClear(SlabPageBegin + i) ? 1 : 0;
@@ -5,23 +5,12 @@
#include "HostAllocator.h"
#include <FEXCore/Utils/MathUtils.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <bitset>
#include <cstddef>
#ifdef TERMUX_BUILD
#ifdef __has_include
#if __has_include(<memory_resource>)
#error Termux <experimental/memory_resource> workaround can be removed
#endif
#endif
#include <experimental/memory_resource>
#include <experimental/list>
namespace fex_pmr = std::experimental::pmr;
#else
#include <memory_resource>
namespace fex_pmr = std::pmr;
#endif
#include <sys/user.h>
#include <mutex>
@@ -88,9 +77,9 @@ namespace Alloc {
IntrusiveArenaAllocator(void* Ptr, size_t _Size)
: Begin {reinterpret_cast<uintptr_t>(Ptr)}
, Size {_Size} {
uint64_t NumberOfPages = _Size / FHU::FEX_PAGE_SIZE;
uint64_t NumberOfPages = _Size / FEXCore::Utils::FEX_PAGE_SIZE;
uint64_t UsedBits = FEXCore::AlignUp(sizeof(IntrusiveArenaAllocator) +
Size / FHU::FEX_PAGE_SIZE / 8, FHU::FEX_PAGE_SIZE);
Size / FEXCore::Utils::FEX_PAGE_SIZE / 8, FEXCore::Utils::FEX_PAGE_SIZE);
for (size_t i = 0; i < UsedBits; ++i) {
UsedPages.Set(i);
}
@@ -118,7 +107,7 @@ namespace Alloc {
void *do_allocate(std::size_t bytes, std::size_t alignment) override {
std::scoped_lock<std::mutex> lk{AllocationMutex};
size_t NumberPages = FEXCore::AlignUp(bytes, FHU::FEX_PAGE_SIZE) / FHU::FEX_PAGE_SIZE;
size_t NumberPages = FEXCore::AlignUp(bytes, FEXCore::Utils::FEX_PAGE_SIZE) / FEXCore::Utils::FEX_PAGE_SIZE;
uintptr_t AllocatedOffset{};
@@ -162,7 +151,7 @@ namespace Alloc {
LastAllocatedPageOffset = AllocatedOffset + NumberPages;
// Now convert this base page to a pointer and return it
return reinterpret_cast<void*>(Begin + AllocatedOffset * FHU::FEX_PAGE_SIZE);
return reinterpret_cast<void*>(Begin + AllocatedOffset * FEXCore::Utils::FEX_PAGE_SIZE);
}
return nullptr;
@@ -171,8 +160,8 @@ namespace Alloc {
void do_deallocate(void* p, std::size_t bytes, std::size_t alignment) override {
std::scoped_lock<std::mutex> lk{AllocationMutex};
uintptr_t PageOffset = (reinterpret_cast<uintptr_t>(p) - Begin) / FHU::FEX_PAGE_SIZE;
size_t NumPages = FEXCore::AlignUp(bytes, FHU::FEX_PAGE_SIZE) / FHU::FEX_PAGE_SIZE;
uintptr_t PageOffset = (reinterpret_cast<uintptr_t>(p) - Begin) / FEXCore::Utils::FEX_PAGE_SIZE;
size_t NumPages = FEXCore::AlignUp(bytes, FEXCore::Utils::FEX_PAGE_SIZE) / FEXCore::Utils::FEX_PAGE_SIZE;
// Walk the allocation list and deallocate
uint64_t FreedPages{};
@@ -1,5 +1,6 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/CPUBackend.h"
#include "Utils/SpinWaitLock.h"
#include <FEXCore/Debug/InternalThreadState.h>
+6
View File
@@ -0,0 +1,6 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/fextl/string.h>
namespace FEXCore::Config {
fextl::string const& GetTelemetryDirectory();
}
+10 -9
View File
@@ -7,6 +7,8 @@
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/Filesystem.h>
#include "Utils/Config.h"
#include <array>
#include <stddef.h>
#include <string_view>
@@ -33,6 +35,7 @@ namespace FEXCore::Telemetry {
"Uses 32-bit Segment SS",
"Uses 32-bit Segment CS",
"Uses 32-bit Segment DS",
"Non-Canonical 64-bit address access",
};
static bool Enabled {true};
@@ -43,8 +46,7 @@ namespace FEXCore::Telemetry {
return;
}
auto DataDirectory = Config::GetDataDirectory();
DataDirectory += "Telemetry/";
auto DataDirectory = Config::GetTelemetryDirectory();
// Ensure the folder structure is created for our configuration
if (!FHU::Filesystem::Exists(DataDirectory) &&
@@ -58,14 +60,13 @@ namespace FEXCore::Telemetry {
return;
}
auto DataDirectory = Config::GetDataDirectory();
DataDirectory += "Telemetry/" + ApplicationName + ".telem";
auto DataDirectory = Config::GetTelemetryDirectory() + ApplicationName + ".telem";
if (FHU::Filesystem::Exists(DataDirectory)) {
// If the file exists, retain a single backup
auto Backup = DataDirectory + ".1";
FHU::Filesystem::CopyFile(DataDirectory, Backup, FHU::Filesystem::CopyOptions::OVERWRITE_EXISTING);
}
// Retain a single backup if the telemetry already existed.
auto Backup = DataDirectory + ".bck";
// Failure on rename is okay.
(void)FHU::Filesystem::RenameFile(DataDirectory, Backup);
auto File = FEXCore::File::File(DataDirectory.c_str(),
FEXCore::File::FileModes::WRITE |
+1 -3
View File
@@ -35,8 +35,6 @@ namespace Handler {
return "1";
else if (Value == "full")
return "2";
else if (Value == "mman")
return "3";
return "0";
}
static inline std::optional<fextl::string> CacheObjectCodeHandler(std::string_view Value) {
@@ -67,7 +65,6 @@ namespace Handler {
CONFIG_SMC_NONE,
CONFIG_SMC_MTRACK,
CONFIG_SMC_FULL,
CONFIG_SMC_MMAN,
};
enum ConfigObjectCodeHandler {
@@ -84,6 +81,7 @@ namespace Handler {
LAYER_GLOBAL_APP,
LAYER_LOCAL_STEAM_APP,
LAYER_LOCAL_APP,
LAYER_USER_OVERRIDE,
LAYER_ENVIRONMENT,
LAYER_TOP,
};
+10
View File
@@ -235,6 +235,16 @@ namespace FEXCore::Context {
FEX_DEFAULT_VISIBILITY virtual void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) = 0;
/**
* @brief Checks if a PC is inside of a thread's JIT code buffer.
*
* @param Thread Which thread's code buffers to check inside of.
* @param Address The PC to check against.
*
* @return true if PC is inside the thread's code buffers.
*/
FEX_DEFAULT_VISIBILITY virtual bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState *Thread, uintptr_t Address) const = 0;
/**
* @brief Allows the frontend to register its own thunk handlers independent of what is controlled in the backend.
*
+9 -4
View File
@@ -1,11 +1,11 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "FEXCore/IR/IR.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/HLE/Linux/ThreadManagement.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/Core/CPUBackend.h>
#include <atomic>
#include <cstddef>
@@ -147,8 +147,13 @@ namespace FEXCore::Core {
}
#endif
flags[1] = 1; ///< Reserved - Always 1.
flags[9] = 1; ///< Interrupt flag - Always 1.
flags[X86State::RFLAG_RESERVED_LOC] = 1; ///< Reserved - Always 1.
flags[X86State::RFLAG_IF_LOC] = 1; ///< Interrupt flag - Always 1.
// DF needs to be initialized to 0 to comply with the Linux ABI. However,
// we encode DF as 1/-1 within the JIT, so we have to write 0x1 here to
// zero DF.
flags[X86State::RFLAG_DF_RAW_LOC] = 0x1;
}
};
static_assert(std::is_trivially_copyable_v<CPUState>, "Needs to be trivial");
+1 -1
View File
@@ -64,7 +64,7 @@ enum X86RegLocation : uint32_t {
RFLAG_SF_RAW_LOC = 7, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
RFLAG_TF_LOC = 8,
RFLAG_IF_LOC = 9,
RFLAG_DF_LOC = 10,
RFLAG_DF_RAW_LOC = 10, // Contains multiple bits, needs to be reconstructed using `ReconstructCompactedEFLAGS`
RFLAG_OF_RAW_LOC = 11, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
RFLAG_IOPL_LOC = 12,
RFLAG_NT_LOC = 14,
@@ -2,17 +2,13 @@
#pragma once
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/InterruptableConditionVariable.h>
#include <FEXCore/Utils/Threads.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/robin_map.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <chrono>
#include <shared_mutex>
@@ -28,6 +24,7 @@ namespace FEXCore::Context {
}
namespace FEXCore::CPU {
class CPUBackend;
union Relocation;
}
@@ -63,14 +60,6 @@ namespace FEXCore::Core {
fextl::vector<FEXCore::CPU::Relocation> *Relocations;
};
struct LocalIREntry {
uint64_t StartAddr;
uint64_t Length;
fextl::unique_ptr<FEXCore::IR::IRListView, FEXCore::IR::IRListViewDeleter> IR;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
fextl::unique_ptr<FEXCore::Core::DebugData> DebugData;
};
// Buffered JIT symbol tracking.
struct JITSymbolBuffer {
// Maximum buffer size to ensure we are a page in size.
@@ -115,8 +104,6 @@ namespace FEXCore::Core {
fextl::unique_ptr<FEXCore::CPU::CPUBackend> CPUBackend;
fextl::unique_ptr<FEXCore::LookupCache> LookupCache;
fextl::robin_map<uint64_t, LocalIREntry> DebugStore;
fextl::unique_ptr<FEXCore::Frontend::Decoder> FrontendDecoder;
fextl::unique_ptr<FEXCore::IR::PassManager> PassManager;
FEXCore::HLE::ThreadManagement ThreadManager;
@@ -146,7 +133,7 @@ namespace FEXCore::Core {
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState{};
// Can be reprotected as RO to trigger an interrupt at generated code block entrypoints
alignas(FHU::FEX_PAGE_SIZE) uint8_t InterruptFaultPage[FHU::FEX_PAGE_SIZE];
alignas(FEXCore::Utils::FEX_PAGE_SIZE) uint8_t InterruptFaultPage[FEXCore::Utils::FEX_PAGE_SIZE];
};
static_assert((offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) -
offsetof(FEXCore::Core::InternalThreadState, BaseFrameState)) < 4096,
@@ -5,10 +5,6 @@
#include <FEXCore/IR/IR.h>
namespace FEXCore {
class CodeLoader;
}
namespace FEXCore::IR {
struct AOTIRCacheEntry;
}
@@ -74,7 +70,6 @@ namespace FEXCore::HLE {
virtual FEXCore::IR::SyscallFlags GetSyscallFlags(uint64_t Syscall) const { return FEXCore::IR::SyscallFlags::DEFAULT; }
SyscallOSABI GetOSABI() const { return OSABI; }
virtual FEXCore::CodeLoader *GetCodeLoader() const { return nullptr; }
virtual void MarkGuestExecutableRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) { }
virtual AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestAddr) = 0;
+13 -633
View File
@@ -8,13 +8,8 @@
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/Utils/EnumOperators.h>
#include <array>
#include <cassert>
#include <cstdint>
#include <cstring>
#include <functional>
#include <memory>
#include <tuple>
#include <fmt/format.h>
@@ -24,477 +19,6 @@ class OrderedNode;
class RegisterAllocationPass;
class RegisterAllocationData;
/**
* @brief The IROp_Header is an dynamically sized array
* At the end it contains a uint8_t for the number of arguments that Op has
* Then there is an unsized array of NodeWrapper arguments for the number of arguments this op has
* The op structures that are including the header must ensure that they pad themselves correctly to the number of arguments used
*/
struct IROp_Header;
/**
* @brief Represents the ID of a given IR node.
*
* Intended to provide strong typing from other integer values
* to prevent passing incorrect values to certain API functions.
*/
struct NodeID final {
using value_type = uint32_t;
constexpr NodeID() noexcept = default;
constexpr explicit NodeID(value_type Value_) noexcept : Value{Value_} {}
constexpr NodeID(const NodeID&) noexcept = default;
constexpr NodeID& operator=(const NodeID&) noexcept = default;
constexpr NodeID(NodeID&&) noexcept = default;
constexpr NodeID& operator=(NodeID&&) noexcept = default;
[[nodiscard]] constexpr bool IsValid() const noexcept {
return Value != 0;
}
[[nodiscard]] constexpr bool IsInvalid() const noexcept {
return !IsValid();
}
constexpr void Invalidate() noexcept {
Value = 0;
}
[[nodiscard]] friend constexpr bool operator==(NodeID, NodeID) noexcept = default;
[[nodiscard]] friend constexpr bool operator<(NodeID lhs, NodeID rhs) noexcept {
return lhs.Value < rhs.Value;
}
[[nodiscard]] friend constexpr bool operator>(NodeID lhs, NodeID rhs) noexcept {
return operator<(rhs, lhs);
}
[[nodiscard]] friend constexpr bool operator<=(NodeID lhs, NodeID rhs) noexcept {
return !operator>(lhs, rhs);
}
[[nodiscard]] friend constexpr bool operator>=(NodeID lhs, NodeID rhs) noexcept {
return !operator<(lhs, rhs);
}
friend std::ostream& operator<<(std::ostream& out, NodeID ID) {
out << ID.Value;
return out;
}
friend std::istream& operator>>(std::istream& in, NodeID& ID) {
in >> ID.Value;
return in;
}
value_type Value{};
};
/**
* @brief This is a very simple wrapper for our node pointers
* You probably don't want to use this directly
* Use OpNodeWrapper and OrderedNodeWrapper types below instead
*
* This is necessary to allow two things
* - Reduce memory usage by having the pointer be an 32bit offset rather than the whole 64bit pointer
* - Actually use an offset from a base so we aren't storing pointers for everything
* - Makes IR list copying be as cheap as a memcpy
* Downsides
* - The IR nodes have to be allocated out of a linear array of memory
* - We currently only allow a 32bit offset, so *only* 4 million nodes per list
* - We have to have the base offset live somewhere else
* - Has to be POD and trivially copyable
* - Makes every real node access turn in to a [Base + Offset] access
* - Can be confusing if you're mixing OpNodeWrapper and OrderedNodeWrapper usage
*/
template<typename Type>
struct NodeWrapperBase final {
// On x86-64 using a uint64_t type is more efficient since RIP addressing gives you [<Base> + <Index> + <imm offset>]
// On AArch64 using uint32_t is just more memory efficient. 32bit or 64bit offset doesn't matter
// We use uint32_t to be more memory efficient (Cuts our node list size in half)
using NodeOffsetType = uint32_t;
NodeOffsetType NodeOffset;
explicit NodeWrapperBase() = default;
[[nodiscard]] static NodeWrapperBase WrapOffset(NodeOffsetType Offset) {
NodeWrapperBase Wrapped;
Wrapped.NodeOffset = Offset;
return Wrapped;
}
[[nodiscard]] static NodeWrapperBase WrapPtr(uintptr_t Base, uintptr_t Value) {
NodeWrapperBase Wrapped;
Wrapped.SetOffset(Base, Value);
return Wrapped;
}
[[nodiscard]] static void *UnwrapNode(uintptr_t Base, NodeWrapperBase Node) {
return Node.GetNode(Base);
}
[[nodiscard]] NodeID ID() const;
[[nodiscard]] bool IsInvalid() const { return NodeOffset == 0; }
[[nodiscard]] Type *GetNode(uintptr_t Base) {
return reinterpret_cast<Type*>(Base + NodeOffset);
}
[[nodiscard]] const Type *GetNode(uintptr_t Base) const {
return reinterpret_cast<const Type*>(Base + NodeOffset);
}
void SetOffset(uintptr_t Base, uintptr_t Value) { NodeOffset = Value - Base; }
[[nodiscard]] friend constexpr bool operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
};
static_assert(std::is_trivial_v<NodeWrapperBase<OrderedNode>>);
static_assert(sizeof(NodeWrapperBase<OrderedNode>) == sizeof(uint32_t));
using OpNodeWrapper = NodeWrapperBase<IROp_Header>;
using OrderedNodeWrapper = NodeWrapperBase<OrderedNode>;
struct OrderedNodeHeader {
OpNodeWrapper Value;
OrderedNodeWrapper Next;
OrderedNodeWrapper Previous;
};
static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3);
/**
* @brief This is a node in our IR representation
* Is a doubly linked list node that lives in a representation of a linearly allocated node list
* The links in the nodes can live in a list independent of the data IR data
*
* ex.
* Region1 : ... <-> <OrderedNode> <-> <OrderedNode> <-> ...
* | *<Value> |
* v v
* Region2 : <IROp>..<IROp>..<IROp>..<IROp>
*
* In this example the OrderedNodes are allocated in one linear memory region (Not necessarily contiguous with one another linking)
* The second region is contiguous but they don't have any relationship with one another directly
*/
class OrderedNode final {
friend class NodeWrapperIterator;
friend class OrderedList;
public:
// These three values are laid out very specifically to make it fast to access the NodeWrappers specifically
OrderedNodeHeader Header;
uint32_t NumUses;
using value_type = OrderedNodeWrapper;
OrderedNode() = default;
/**
* @brief Appends a node to this current node
*
* Before. <Prev> <-> <Current> <-> <Next>
* After. <Prev> <-> <Current> <-> <Node> <-> Next
*
* @return Pointer to the node being added
*/
value_type append(uintptr_t Base, value_type Node) {
// Set Next Node's Previous to incoming node
SetPrevious(Base, Header.Next, Node);
// Set Incoming node's links to this node's links
SetPrevious(Base, Node, Wrapped(Base));
SetNext(Base, Node, Header.Next);
// Set this node's next to the incoming node
SetNext(Base, Wrapped(Base), Node);
// Return the node we are appending
return Node;
}
OrderedNode *append(uintptr_t Base, OrderedNode *Node) {
value_type WNode = Node->Wrapped(Base);
// Set Next Node's Previous to incoming node
SetPrevious(Base, Header.Next, WNode);
// Set Incoming node's links to this node's links
SetPrevious(Base, WNode, Wrapped(Base));
SetNext(Base, WNode, Header.Next);
// Set this node's next to the incoming node
SetNext(Base, Wrapped(Base), WNode);
// Return the node we are appending
return Node;
}
/**
* @brief Prepends a node to the current node
* Before. <Prev> <-> <Current> <-> <Next>
* After. <Prev> <-> <Node> <-> <Current> <-> Next
*
* @return Pointer to the node being added
*/
value_type prepend(uintptr_t Base, value_type Node) {
// Set the previous node's next to the incoming node
SetNext(Base, Header.Previous, Node);
// Set the incoming node's links
SetPrevious(Base, Node, Header.Previous);
SetNext(Base, Node, Wrapped(Base));
// Set the current node's link
SetPrevious(Base, Wrapped(Base), Node);
// Return the node we are prepending
return Node;
}
OrderedNode *prepend(uintptr_t Base, OrderedNode *Node) {
value_type WNode = Node->Wrapped(Base);
// Set the previous node's next to the incoming node
SetNext(Base, Header.Previous, WNode);
// Set the incoming node's links
SetPrevious(Base, WNode, Header.Previous);
SetNext(Base, WNode, Wrapped(Base));
// Set the current node's link
SetPrevious(Base, Wrapped(Base), WNode);
// Return the node we are prepending
return Node;
}
/**
* @brief Gets the remaining size of the blocks from this point onward
*
* Doesn't find the head of the list
*
*/
[[nodiscard]] size_t size(uintptr_t Base) const {
size_t Size = 1;
// Walk the list forward until we hit a sentinel
value_type Current = Header.Next;
while (Current.NodeOffset != 0) {
++Size;
OrderedNode *RealNode = Current.GetNode(Base);
Current = RealNode->Header.Next;
}
return Size;
}
void Unlink(uintptr_t Base) {
// This removes the node from the list. Orphaning it
// Before: <Previous> <-> <Current> <-> <Next>
// After: <Previous <-> <Next>
SetNext(Base, Header.Previous, Header.Next);
SetPrevious(Base, Header.Next, Header.Previous);
}
[[nodiscard]] IROp_Header const* Op(uintptr_t Base) const {
return Header.Value.GetNode(Base);
}
[[nodiscard]] IROp_Header *Op(uintptr_t Base) {
return Header.Value.GetNode(Base);
}
[[nodiscard]] uint32_t GetUses() const { return NumUses; }
void AddUse() { ++NumUses; }
void RemoveUse() { --NumUses; }
[[nodiscard]] value_type Wrapped(uintptr_t Base) const {
value_type Tmp;
Tmp.SetOffset(Base, reinterpret_cast<uintptr_t>(this));
return Tmp;
}
private:
[[nodiscard]] value_type WrappedOffset(uint32_t Offset) const {
value_type Tmp;
Tmp.NodeOffset = Offset;
return Tmp;
}
static void SetPrevious(uintptr_t Base, value_type Node, value_type New) {
OrderedNode *RealNode = Node.GetNode(Base);
RealNode->Header.Previous = New;
}
static void SetNext(uintptr_t Base, value_type Node, value_type New) {
OrderedNode *RealNode = Node.GetNode(Base);
RealNode->Header.Next = New;
}
void SetUses(uint32_t Uses) { NumUses = Uses; }
};
static_assert(std::is_trivial_v<OrderedNode>);
static_assert(std::is_trivially_copyable_v<OrderedNode>);
static_assert(offsetof(OrderedNode, Header) == 0);
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
struct RegisterClassType final {
using value_type = uint32_t;
value_type Val;
[[nodiscard]] constexpr operator value_type() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
};
struct CondClassType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const CondClassType&, const CondClassType&) = default;
};
struct MemOffsetType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const MemOffsetType&, const MemOffsetType&) = default;
};
struct TypeDefinition final {
uint16_t Val;
[[nodiscard]] constexpr operator uint16_t() const {
return Val;
}
[[nodiscard]] static constexpr TypeDefinition Create(uint8_t Bytes) {
TypeDefinition Type{};
Type.Val = Bytes << 8;
return Type;
}
[[nodiscard]] static constexpr TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
TypeDefinition Type{};
Type.Val = (Bytes << 8) | (Elements & 255);
return Type;
}
[[nodiscard]] constexpr uint8_t Bytes() const {
return Val >> 8;
}
[[nodiscard]] constexpr uint8_t Elements() const {
return Val & 255;
}
[[nodiscard]] friend constexpr bool operator==(const TypeDefinition&, const TypeDefinition&) = default;
};
static_assert(std::is_trivial_v<TypeDefinition>);
struct FenceType final {
using value_type = uint8_t;
value_type Val;
[[nodiscard]] constexpr operator value_type() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
};
struct RoundType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const RoundType&, const RoundType&) = default;
};
struct SHA256Sum final {
uint8_t data[32];
[[nodiscard]] bool operator<(SHA256Sum const &rhs) const {
return memcmp(data, rhs.data, sizeof(data)) < 0;
}
[[nodiscard]] bool operator==(SHA256Sum const &rhs) const {
return memcmp(data, rhs.data, sizeof(data)) == 0;
}
};
typedef void ThunkedFunction(void* ArgsRv);
struct ThunkDefinition final {
SHA256Sum Sum;
ThunkedFunction *ThunkFunction;
};
class NodeIterator;
/* This iterator can be used to step though nodes.
* Due to how our IR is laid out, this can be used to either step
* though the CodeBlocks or though the code within a single block.
*/
class NodeIterator {
public:
using value_type = std::tuple<OrderedNode*, IROp_Header*>;
using size_type = std::size_t;
using difference_type = std::ptrdiff_t;
using reference = value_type&;
using const_reference = const value_type&;
using pointer = value_type*;
using const_pointer = const value_type*;
using iterator = NodeIterator;
using const_iterator = const NodeIterator;
using reverse_iterator = iterator;
using const_reverse_iterator = const_iterator;
using iterator_category = std::bidirectional_iterator_tag;
NodeIterator(uintptr_t Base, uintptr_t IRBase) : BaseList {Base}, IRList{ IRBase } {}
explicit NodeIterator(uintptr_t Base, uintptr_t IRBase, OrderedNodeWrapper Ptr) : BaseList {Base}, IRList{ IRBase }, Node {Ptr} {}
[[nodiscard]] bool operator==(const NodeIterator &rhs) const {
return Node.NodeOffset == rhs.Node.NodeOffset;
}
[[nodiscard]] bool operator!=(const NodeIterator &rhs) const {
return !operator==(rhs);
}
NodeIterator operator++() {
OrderedNodeHeader *RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
Node = RealNode->Next;
return *this;
}
NodeIterator operator--() {
OrderedNodeHeader *RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
Node = RealNode->Previous;
return *this;
}
[[nodiscard]] value_type operator*() {
OrderedNode *RealNode = Node.GetNode(BaseList);
return { RealNode, RealNode->Op(IRList) };
}
[[nodiscard]] value_type operator()() {
OrderedNode *RealNode = Node.GetNode(BaseList);
return { RealNode, RealNode->Op(IRList) };
}
[[nodiscard]] NodeID ID() const {
return Node.ID();
}
[[nodiscard]] static NodeIterator Invalid() {
return NodeIterator(0, 0);
}
protected:
uintptr_t BaseList{};
uintptr_t IRList{};
OrderedNodeWrapper Node{};
};
enum class SyscallFlags : uint8_t {
DEFAULT = 0,
// Syscalldoesn't care about CPUState being serialized up to the syscall instruction.
@@ -533,6 +57,8 @@ enum NamedVectorConstant : uint8_t {
NAMED_VECTOR_BLENDPS_1011B,
NAMED_VECTOR_BLENDPS_1101B,
NAMED_VECTOR_BLENDPS_1110B,
NAMED_VECTOR_MOVMASKB,
NAMED_VECTOR_MOVMASKB_UPPER,
NAMED_VECTOR_CONST_POOL_MAX,
// Beginning of named constants that don't have a constant pool backing.
NAMED_VECTOR_ZERO = NAMED_VECTOR_CONST_POOL_MAX,
@@ -553,168 +79,22 @@ enum IndexNamedVectorConstant : uint8_t {
INDEXED_NAMED_VECTOR_MAX,
};
// This must directly match bytes to the named opsize.
// Implicit sized IR operations does math to get between sizes.
enum OpSize : uint8_t {
i8Bit = 1,
i16Bit = 2,
i32Bit = 4,
i64Bit = 8,
i128Bit = 16,
i256Bit = 32,
};
enum class FloatCompareOp : uint8_t {
EQ = 0,
LT,
LE,
UNO,
NEQ,
ORD,
};
enum class ShiftType : uint8_t {
LSL = 0,
LSR,
ASR,
ROR,
};
// Converts a size stored as an integer in to an OpSize enum.
// This is a nop operation and will be eliminated by the compiler.
static inline OpSize SizeToOpSize(uint8_t Size) {
switch (Size) {
case 1: return OpSize::i8Bit;
case 2: return OpSize::i16Bit;
case 4: return OpSize::i32Bit;
case 8: return OpSize::i64Bit;
case 16: return OpSize::i128Bit;
case 32: return OpSize::i256Bit;
default: FEX_UNREACHABLE;
}
}
#define IROP_ENUM
#define IROP_STRUCTS
#define IROP_SIZES
#define IROP_REG_CLASSES
#include <FEXCore/IR/IRDefines.inc>
/* This iterator can be used to step though every single node in a multi-block in SSA order.
*
* Iterates in the order of:
*
* end <-- CodeBlockA <--> BlockAInst1 <--> BlockAInst2 <--> CodeBlockB <--> BlockBInst1 <--> BlockBInst2 --> end
*/
class AllNodesIterator : public NodeIterator {
public:
AllNodesIterator(uintptr_t Base, uintptr_t IRBase) : NodeIterator(Base, IRBase) {}
explicit AllNodesIterator(uintptr_t Base, uintptr_t IRBase, OrderedNodeWrapper Ptr) : NodeIterator(Base, IRBase, Ptr) {}
AllNodesIterator(NodeIterator other) : NodeIterator(other) {} // Allow NodeIterator to be upgraded
AllNodesIterator operator++() {
OrderedNodeHeader *RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
auto IROp = Node.GetNode(BaseList)->Op(IRList);
// If this is the last node of a codeblock, we need to continue to the next block
if (IROp->Op == OP_ENDBLOCK) {
auto EndBlock = IROp->C<IROp_EndBlock>();
auto CurrentBlock = EndBlock->BlockHeader.GetNode(BaseList);
Node = CurrentBlock->Header.Next;
} else if (IROp->Op == OP_CODEBLOCK) {
auto CodeBlock = IROp->C<IROp_CodeBlock>();
Node = CodeBlock->Begin;
} else {
Node = RealNode->Next;
}
return *this;
struct SHA256Sum final {
uint8_t data[32];
[[nodiscard]] bool operator<(SHA256Sum const &rhs) const {
return memcmp(data, rhs.data, sizeof(data)) < 0;
}
AllNodesIterator operator--() {
auto IROp = Node.GetNode(BaseList)->Op(IRList);
if (IROp->Op == OP_BEGINBLOCK) {
auto BeginBlock = IROp->C<IROp_EndBlock>();
Node = BeginBlock->BlockHeader;
} else if (IROp->Op == OP_CODEBLOCK) {
auto PrevBlockWrapper = Node.GetNode(BaseList)->Header.Previous;
auto PrevCodeBlock = PrevBlockWrapper.GetNode(BaseList)->Op(IRList)->C<IROp_CodeBlock>();
Node = PrevCodeBlock->Last;
} else {
Node = Node.GetNode(BaseList)->Header.Previous;
}
return *this;
}
[[nodiscard]] static AllNodesIterator Invalid() {
return AllNodesIterator(0, 0);
[[nodiscard]] bool operator==(SHA256Sum const &rhs) const {
return memcmp(data, rhs.data, sizeof(data)) == 0;
}
};
class IRListView;
class IREmitter;
typedef void ThunkedFunction(void* ArgsRv);
template<typename Type>
inline NodeID NodeWrapperBase<Type>::ID() const {
return NodeID(NodeOffset / sizeof(IR::OrderedNode));
}
bool IsFragmentExit(FEXCore::IR::IROps Op);
bool IsBlockExit(FEXCore::IR::IROps Op);
struct ThunkDefinition final {
SHA256Sum Sum;
ThunkedFunction *ThunkFunction;
};
} // namespace FEXCore::IR
template <>
struct std::hash<FEXCore::IR::NodeID> {
size_t operator()(const FEXCore::IR::NodeID& ID) const noexcept {
return std::hash<FEXCore::IR::NodeID::value_type>{}(ID.Value);
}
};
template <>
struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID::value_type> {
using Base = fmt::formatter<FEXCore::IR::NodeID::value_type>;
// Pass-through the underlying value, so IDs can
// be formatted like any integral value.
template <typename FormatContext>
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
return Base::format(ID.Value, ctx);
}
};
template <>
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
template <typename FormatContext>
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
return Base::format(Class.Val, ctx);
}
};
template <>
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
template <typename FormatContext>
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
return Base::format(Fence.Val, ctx);
}
};
template <>
struct fmt::formatter<FEXCore::IR::OpSize> : fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>> {
using Base = fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>>;
template <typename FormatContext>
auto format(const FEXCore::IR::OpSize& OpSize, FormatContext& ctx) const {
return Base::format(FEXCore::ToUnderlying(OpSize), ctx);
}
};
@@ -6,6 +6,7 @@
#include <cstdint>
#include <functional>
#include <optional>
#include <sys/types.h>
namespace FEXCore::Allocator {
@@ -72,6 +73,7 @@ namespace FEXCore::Allocator {
size_t Size;
};
FEX_DEFAULT_VISIBILITY fextl::vector<MemoryRegion> CollectMemoryGaps(uintptr_t Begin, uintptr_t End, int MapsFD);
FEX_DEFAULT_VISIBILITY fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End);
FEX_DEFAULT_VISIBILITY void ReclaimMemoryRegion(const fextl::vector<MemoryRegion> & Regions);
// When running a 64-bit executable on ARM then userspace guest only gets 47 bits of VA
@@ -30,6 +30,7 @@ namespace FEXCore::Telemetry {
TYPE_USES_32BIT_SEGMENT_SS,
TYPE_USES_32BIT_SEGMENT_CS,
TYPE_USES_32BIT_SEGMENT_DS,
TYPE_UNHANDLED_NONCANONICAL_ADDRESS,
TYPE_LAST,
};
@@ -2,7 +2,7 @@
#pragma once
#include <cstddef>
namespace FHU {
namespace FEXCore::Utils {
// FEX assumes an operating page size of 4096
// To work around build systems that build on a 16k/64k page size, define our page size here
// Don't use the system provided PAGE_SIZE define because of this.
+1 -1
View File
@@ -1,5 +1,5 @@
#include <FEXCore/Utils/FileLoading.h>
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
TEST_CASE("LoadFile-Doesn'tExist") {
fextl::string MapsFile;
+1 -1
View File
@@ -1,5 +1,5 @@
#include "Utils/SpinWaitLock.h"
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
#include <chrono>
#include <thread>
+2 -1
View File
@@ -1,5 +1,6 @@
#include <FEXCore/Utils/MathUtils.h>
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
#include <catch2/generators/catch_generators_range.hpp>
TEST_CASE("ILog2") {
auto i = GENERATE(range(0, 64));
+2 -2
View File
@@ -1,6 +1,6 @@
#include "TestDisassembler.h"
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
#include <fcntl.h>
using namespace FEXCore::ARMEmitter;
@@ -450,7 +450,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Bitfield") {
TEST_SINGLE(bfxil(Size::i32Bit, Reg::r29, Reg::r28, 4, 3), "bfxil w29, w28, #4, #3");
TEST_SINGLE(bfxil(Size::i32Bit, Reg::r29, Reg::r28, 27, 3), "bfxil w29, w28, #27, #3");
TEST_SINGLE(bfxil(Size::i64Bit, Reg::r29, Reg::r28, 4, 3), "bfxil x29, x28, #4, #3");
TEST_SINGLE(bfxil(Size::i64Bit, Reg::r29, Reg::r28, 57, 3), "bfxil x29, x28, #57, #3");
+1 -1
View File
@@ -1,6 +1,6 @@
#include "TestDisassembler.h"
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
#include <fcntl.h>
using namespace FEXCore::ARMEmitter;
+1 -1
View File
@@ -1,6 +1,6 @@
#include "TestDisassembler.h"
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
#include <fcntl.h>
using namespace FEXCore::ARMEmitter;
@@ -1,6 +1,6 @@
#include "TestDisassembler.h"
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
#include <fcntl.h>
using namespace FEXCore::ARMEmitter;
+12 -12
View File
@@ -1,6 +1,6 @@
#include "TestDisassembler.h"
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
#include <fcntl.h>
using namespace FEXCore::ARMEmitter;
@@ -1177,7 +1177,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE inc/dec vector by element
TEST_SINGLE(inch(ZReg::z30, PredicatePattern::SVE_POW2 , 1), "inch z30.h, pow2");
TEST_SINGLE(inch(ZReg::z30, PredicatePattern::SVE_VL256, 7), "inch z30.h, vl256, mul #7");
TEST_SINGLE(inch(ZReg::z30, PredicatePattern::SVE_ALL , 16), "inch z30.h, all, mul #16");
TEST_SINGLE(dech(ZReg::z30, PredicatePattern::SVE_POW2 , 1), "dech z30.h, pow2");
TEST_SINGLE(dech(ZReg::z30, PredicatePattern::SVE_VL256, 7), "dech z30.h, vl256, mul #7");
TEST_SINGLE(dech(ZReg::z30, PredicatePattern::SVE_ALL , 16), "dech z30.h, all, mul #16");
@@ -1185,7 +1185,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE inc/dec vector by element
TEST_SINGLE(incw(ZReg::z30, PredicatePattern::SVE_POW2 , 1), "incw z30.s, pow2");
TEST_SINGLE(incw(ZReg::z30, PredicatePattern::SVE_VL256, 7), "incw z30.s, vl256, mul #7");
TEST_SINGLE(incw(ZReg::z30, PredicatePattern::SVE_ALL , 16), "incw z30.s, all, mul #16");
TEST_SINGLE(decw(ZReg::z30, PredicatePattern::SVE_POW2 , 1), "decw z30.s, pow2");
TEST_SINGLE(decw(ZReg::z30, PredicatePattern::SVE_VL256, 7), "decw z30.s, vl256, mul #7");
TEST_SINGLE(decw(ZReg::z30, PredicatePattern::SVE_ALL , 16), "decw z30.s, all, mul #16");
@@ -1193,7 +1193,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE inc/dec vector by element
TEST_SINGLE(incd(ZReg::z30, PredicatePattern::SVE_POW2 , 1), "incd z30.d, pow2");
TEST_SINGLE(incd(ZReg::z30, PredicatePattern::SVE_VL256, 7), "incd z30.d, vl256, mul #7");
TEST_SINGLE(incd(ZReg::z30, PredicatePattern::SVE_ALL , 16), "incd z30.d, all, mul #16");
TEST_SINGLE(decd(ZReg::z30, PredicatePattern::SVE_POW2 , 1), "decd z30.d, pow2");
TEST_SINGLE(decd(ZReg::z30, PredicatePattern::SVE_VL256, 7), "decd z30.d, vl256, mul #7");
TEST_SINGLE(decd(ZReg::z30, PredicatePattern::SVE_ALL , 16), "decd z30.d, all, mul #16");
@@ -1203,7 +1203,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE inc/dec register by elemen
TEST_SINGLE(incb(XReg::x30, PredicatePattern::SVE_POW2 , 1), "incb x30, pow2");
TEST_SINGLE(incb(XReg::x30, PredicatePattern::SVE_VL256, 7), "incb x30, vl256, mul #7");
TEST_SINGLE(incb(XReg::x30, PredicatePattern::SVE_ALL , 16), "incb x30, all, mul #16");
TEST_SINGLE(decb(XReg::x30, PredicatePattern::SVE_POW2 , 1), "decb x30, pow2");
TEST_SINGLE(decb(XReg::x30, PredicatePattern::SVE_VL256, 7), "decb x30, vl256, mul #7");
TEST_SINGLE(decb(XReg::x30, PredicatePattern::SVE_ALL , 16), "decb x30, all, mul #16");
@@ -1211,7 +1211,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE inc/dec register by elemen
TEST_SINGLE(inch(XReg::x30, PredicatePattern::SVE_POW2 , 1), "inch x30, pow2");
TEST_SINGLE(inch(XReg::x30, PredicatePattern::SVE_VL256, 7), "inch x30, vl256, mul #7");
TEST_SINGLE(inch(XReg::x30, PredicatePattern::SVE_ALL , 16), "inch x30, all, mul #16");
TEST_SINGLE(dech(XReg::x30, PredicatePattern::SVE_POW2 , 1), "dech x30, pow2");
TEST_SINGLE(dech(XReg::x30, PredicatePattern::SVE_VL256, 7), "dech x30, vl256, mul #7");
TEST_SINGLE(dech(XReg::x30, PredicatePattern::SVE_ALL , 16), "dech x30, all, mul #16");
@@ -1219,7 +1219,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE inc/dec register by elemen
TEST_SINGLE(incw(XReg::x30, PredicatePattern::SVE_POW2 , 1), "incw x30, pow2");
TEST_SINGLE(incw(XReg::x30, PredicatePattern::SVE_VL256, 7), "incw x30, vl256, mul #7");
TEST_SINGLE(incw(XReg::x30, PredicatePattern::SVE_ALL , 16), "incw x30, all, mul #16");
TEST_SINGLE(decw(XReg::x30, PredicatePattern::SVE_POW2 , 1), "decw x30, pow2");
TEST_SINGLE(decw(XReg::x30, PredicatePattern::SVE_VL256, 7), "decw x30, vl256, mul #7");
TEST_SINGLE(decw(XReg::x30, PredicatePattern::SVE_ALL , 16), "decw x30, all, mul #16");
@@ -1227,7 +1227,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE inc/dec register by elemen
TEST_SINGLE(incd(XReg::x30, PredicatePattern::SVE_POW2 , 1), "incd x30, pow2");
TEST_SINGLE(incd(XReg::x30, PredicatePattern::SVE_VL256, 7), "incd x30, vl256, mul #7");
TEST_SINGLE(incd(XReg::x30, PredicatePattern::SVE_ALL , 16), "incd x30, all, mul #16");
TEST_SINGLE(decd(XReg::x30, PredicatePattern::SVE_POW2 , 1), "decd x30, pow2");
TEST_SINGLE(decd(XReg::x30, PredicatePattern::SVE_VL256, 7), "decd x30, vl256, mul #7");
TEST_SINGLE(decd(XReg::x30, PredicatePattern::SVE_ALL , 16), "decd x30, all, mul #16");
@@ -1507,7 +1507,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE Permute Predicate") {
TEST_SINGLE(rev(SubRegSize::i16Bit, PReg::p15, PReg::p14), "rev p15.h, p14.h");
TEST_SINGLE(rev(SubRegSize::i32Bit, PReg::p15, PReg::p14), "rev p15.s, p14.s");
TEST_SINGLE(rev(SubRegSize::i64Bit, PReg::p15, PReg::p14), "rev p15.d, p14.d");
TEST_SINGLE(punpklo(PReg::p15, PReg::p14), "punpklo p15.h, p14.b");
TEST_SINGLE(punpkhi(PReg::p15, PReg::p14), "punpkhi p15.h, p14.b");
@@ -3469,15 +3469,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE floating-point multiply-ad
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalb z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalb z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalb z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalt z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalt z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlalt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlalt z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslb z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslb z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlalb(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslb z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlslt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslt z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlslt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslt z30.s, z29.h, z28.h");
// TEST_SINGLE(bfmlslt(SubRegSize::i32Bit, ZReg::z30, ZReg::z29, ZReg::z28), "bfmlslt z30.s, z29.h, z28.h");
+1 -1
View File
@@ -1,6 +1,6 @@
#include "TestDisassembler.h"
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
#include <fcntl.h>
using namespace FEXCore::ARMEmitter;
+1 -1
View File
@@ -1,6 +1,6 @@
#include "TestDisassembler.h"
#include <catch2/catch.hpp>
#include <catch2/catch_test_macros.hpp>
#include <fcntl.h>
using namespace FEXCore::ARMEmitter;
-6
View File
@@ -60,12 +60,6 @@ DefinitionRenameDict = {
"pread64": "pread_64",
"pwrite64": "pwrite_64",
"prlimit64": "prlimit_64",
# Shm symbols conflict with termux defines and FEX's syscall token pasting.
# Underscore at the start to avoid name collision
"shmget": "_shmget",
"shmctl": "_shmctl",
"shmat": "_shmat",
"shmdt": "_shmdt",
# musl/Alpine Linux defines `fstatat64` as a define that points to `fstatat`.
# Rename it to avoid global define conflicts.
"fstatat64": "fstatat_64",
+18 -15
View File
@@ -129,6 +129,8 @@ namespace JSON {
public:
explicit MainLoader(FEXCore::Config::LayerType Type);
explicit MainLoader(fextl::string ConfigFile);
explicit MainLoader(FEXCore::Config::LayerType Type, const char* ConfigFile);
void Load() override;
private:
@@ -188,6 +190,12 @@ namespace JSON {
, Config{std::move(ConfigFile)} {
}
MainLoader::MainLoader(FEXCore::Config::LayerType Type, const char* ConfigFile)
: OptionMapper(Type)
, Config{ConfigFile} {
}
void MainLoader::Load() {
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
MapNameToOption(Name, ConfigString);
@@ -276,6 +284,10 @@ namespace JSON {
}
}
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(const char* AppConfig) {
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_USER_OVERRIDE, AppConfig);
}
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type) {
return fextl::make_unique<AppLoader>(Filename, Type);
}
@@ -371,6 +383,11 @@ namespace JSON {
FEXCore::Config::AddLayer(fextl::make_unique<FEX::ArgLoader::ArgLoader>(argc, argv));
}
const char *AppConfig = getenv("FEX_APP_CONFIG");
if (AppConfig && FHU::Filesystem::Exists(AppConfig)) {
FEXCore::Config::AddLayer(CreateUserOverrideLayer(AppConfig));
}
FEXCore::Config::AddLayer(CreateEnvironmentLayer(envp));
FEXCore::Config::Load();
@@ -529,21 +546,7 @@ namespace JSON {
}
fextl::string GetConfigFileLocation(bool Global) {
fextl::string ConfigFile{};
if (Global) {
ConfigFile = GetConfigDirectory(true) + "Config.json";
}
else {
const char *AppConfig = getenv("FEX_APP_CONFIG");
if (AppConfig) {
// App config environment variable overwrites only the config file
ConfigFile = AppConfig;
}
else {
ConfigFile = GetConfigDirectory(false) + "Config.json";
}
}
return ConfigFile;
return GetConfigDirectory(Global) + "Config.json";
}
void InitializeConfigs() {
+1
View File
@@ -73,6 +73,7 @@ namespace FEX::Config {
* @return unique_ptr for that layer
*/
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File = nullptr);
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(const char* AppConfig);
/**
* @brief Create an application configuration loader
+1 -4
View File
@@ -5,15 +5,12 @@ if (NOT MINGW_BUILD)
add_subdirectory(FEXConfig/)
endif()
if (NOT TERMUX_BUILD)
# Disable FEXRootFSFetcher on Termux, it doesn't even work there
add_subdirectory(FEXRootFSFetcher/)
endif()
if (ENABLE_GDB_SYMBOLS)
add_subdirectory(FEXGDBReader/)
endif()
add_subdirectory(FEXRootFSFetcher/)
add_subdirectory(FEXGetConfig/)
add_subdirectory(FEXServer/)
add_subdirectory(FEXBash/)
@@ -6,11 +6,12 @@
#include <cstdint>
#include <functional>
namespace FEXCore {
namespace IR {
namespace FEXCore::IR {
class IREmitter;
}
namespace FEX {
/**
* @brief Code loader class so the CPU backend can load code in a generic fashion
*
@@ -46,5 +47,4 @@ public:
virtual uint64_t GetBaseOffset() const { return 0; }
};
}
+9 -12
View File
@@ -1,6 +1,7 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "CodeLoader.h"
#include "Common/Config.h"
#include <array>
@@ -9,7 +10,6 @@
#include <cstring>
#include <fcntl.h>
#include <FEXCore/Core/CodeLoader.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Utils/Allocator.h>
@@ -17,13 +17,13 @@
#include <FEXCore/Utils/FileLoading.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <unistd.h>
namespace FEX::HarnessHelper {
@@ -125,7 +125,7 @@ namespace FEX::HarnessHelper {
}
if (BaseConfig.OptionRegDataCount > 0) {
static constexpr std::array<uint64_t, 44> OffsetArrayAVX = {{
static constexpr std::array<uint64_t, 43> OffsetArrayAVX = {{
offsetof(FEXCore::Core::CPUState, rip),
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]),
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RBX]),
@@ -169,9 +169,8 @@ namespace FEX::HarnessHelper {
offsetof(FEXCore::Core::CPUState, mm[5][0]),
offsetof(FEXCore::Core::CPUState, mm[6][0]),
offsetof(FEXCore::Core::CPUState, mm[7][0]),
offsetof(FEXCore::Core::CPUState, mm[8][0]),
}};
static constexpr std::array<uint64_t, 44> OffsetArraySSE = {{
static constexpr std::array<uint64_t, 43> OffsetArraySSE = {{
offsetof(FEXCore::Core::CPUState, rip),
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]),
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RBX]),
@@ -215,7 +214,6 @@ namespace FEX::HarnessHelper {
offsetof(FEXCore::Core::CPUState, mm[5][0]),
offsetof(FEXCore::Core::CPUState, mm[6][0]),
offsetof(FEXCore::Core::CPUState, mm[7][0]),
offsetof(FEXCore::Core::CPUState, mm[8][0]),
}};
uintptr_t DataOffset = BaseConfig.OptionRegDataOffset;
@@ -254,10 +252,9 @@ namespace FEX::HarnessHelper {
Name = "gs";
else if (NameIndex == 34)
Name ="fs";
else if (NameIndex == 35)
Name = "rflags";
else if (NameIndex >= 36 && NameIndex < 45)
Name = fextl::fmt::format("MM[{}][{}]", NameIndex - 36, j);
else if (NameIndex >= 35 && NameIndex < 43) {
Name = fextl::fmt::format("MM[{}][{}]", NameIndex - 35, j);
}
if (State1) {
CheckGPRs(fextl::fmt::format("Core1: {}: ", Name), State1Data[j], RegData->RegValues[j]);
@@ -369,7 +366,7 @@ namespace FEX::HarnessHelper {
ConfigStructBase BaseConfig;
};
class HarnessCodeLoader final : public FEXCore::CodeLoader {
class HarnessCodeLoader final : public FEX::CodeLoader {
public:
HarnessCodeLoader(fextl::string const &Filename, fextl::string const &ConfigFilename) {
@@ -425,7 +422,7 @@ namespace FEX::HarnessHelper {
// Map in the memory region for the test file
#ifndef _WIN32
size_t Length = FEXCore::AlignUp(RawASMFile.size(), FHU::FEX_PAGE_SIZE);
size_t Length = FEXCore::AlignUp(RawASMFile.size(), FEXCore::Utils::FEX_PAGE_SIZE);
auto ASMPtr = FEXCore::Allocator::VirtualAlloc(reinterpret_cast<void*>(Code_start_page), Length, true);
#else
// Special magic DOS area that starts at 0x1'0000
@@ -146,9 +146,9 @@ struct ELFParser {
}
if (type == ::ELFLoader::ELFContainer::TYPE_X86_32) {
Elf32_Phdr phdrs32[ehdr.e_phnum];
fextl::vector<Elf32_Phdr> phdrs32(ehdr.e_phnum);
if (pread(fd, phdrs32, sizeof(Elf32_Phdr) * ehdr.e_phnum, ehdr.e_phoff) == -1) {
if (pread(fd, phdrs32.data(), sizeof(Elf32_Phdr) * ehdr.e_phnum, ehdr.e_phoff) == -1) {
LogMan::Msg::EFmt("Failed to read phdr32 from '{}'", fd);
return false;
}
+33 -4
View File
@@ -547,12 +547,44 @@ namespace {
void FillHackConfig() {
if (ImGui::BeginTabItem("Hacks")) {
auto Value = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED);
auto VectorTSO = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED);
auto MemcpyTSO = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED);
bool TSOEnabled = Value.has_value() && **Value == "1";
bool VectorTSOEnabled = VectorTSO.has_value() && **VectorTSO == "1";
bool MemcpyTSOEnabled = MemcpyTSO.has_value() && **MemcpyTSO == "1";
if (ImGui::Checkbox("TSO Enabled", &TSOEnabled)) {
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, TSOEnabled ? "1" : "0");
ConfigChanged = true;
}
if (TSOEnabled) {
if (ImGui::TreeNodeEx("TSO sub-options", ImGuiTreeNodeFlags_Leaf)) {
if (ImGui::Checkbox("Vector TSO Enabled", &VectorTSOEnabled)) {
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, VectorTSOEnabled ? "1" : "0");
ConfigChanged = true;
}
if (ImGui::IsItemHovered()) {
ImGui::BeginTooltip();
ImGui::Text("Disables TSO emulation on vector load/store instructions");
ImGui::EndTooltip();
}
if (ImGui::Checkbox("Memcpy TSO Enabled", &MemcpyTSOEnabled)) {
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, MemcpyTSOEnabled ? "1" : "0");
ConfigChanged = true;
}
if (ImGui::IsItemHovered()) {
ImGui::BeginTooltip();
ImGui::Text("Disables TSO emulation on memcpy/memset instructions");
ImGui::EndTooltip();
}
ImGui::TreePop();
}
}
Value = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_PARANOIDTSO);
bool ParanoidTSOEnabled = Value.has_value() && **Value == "1";
if (ImGui::Checkbox("Paranoid TSO Enabled", &ParanoidTSOEnabled)) {
@@ -568,7 +600,7 @@ namespace {
}
ImGui::Text("SMC Checks: ");
int SMCChecks = FEXCore::Config::CONFIG_SMC_MMAN;
int SMCChecks = FEXCore::Config::CONFIG_SMC_MTRACK;
Value = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_SMCCHECKS);
if (Value.has_value()) {
@@ -578,8 +610,6 @@ namespace {
SMCChecks = FEXCore::Config::CONFIG_SMC_MTRACK;
} else if (**Value == "2") {
SMCChecks = FEXCore::Config::CONFIG_SMC_FULL;
} else if (**Value == "3") {
SMCChecks = FEXCore::Config::CONFIG_SMC_MMAN;
}
}
@@ -587,7 +617,6 @@ namespace {
SMCChanged |= ImGui::RadioButton("None", &SMCChecks, FEXCore::Config::CONFIG_SMC_NONE); ImGui::SameLine();
SMCChanged |= ImGui::RadioButton("MTrack (Default)", &SMCChecks, FEXCore::Config::CONFIG_SMC_MTRACK); ImGui::SameLine();
SMCChanged |= ImGui::RadioButton("Full", &SMCChecks, FEXCore::Config::CONFIG_SMC_FULL);
SMCChanged |= ImGui::RadioButton("MMan (Deprecated)", &SMCChecks, FEXCore::Config::CONFIG_SMC_MMAN); ImGui::SameLine();
if (SMCChanged) {
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_SMCCHECKS, std::to_string(SMCChecks));
-5
View File
@@ -5,11 +5,6 @@ if (ENABLE_VIXL_SIMULATOR)
list(APPEND DEFINES -DVIXL_SIMULATOR=1)
endif()
if (TERMUX_BUILD)
# Termux needs android-shmem to get the shm emulation library.
list(APPEND LIBS android-shmem)
endif()
function(GenerateInterpreter NAME AsInterpreter)
add_executable(${NAME}
FEXLoader.cpp
+7 -7
View File
@@ -3,6 +3,7 @@
#pragma once
#include "ArchHelpers/UContext.h"
#include "CodeLoader.h"
#include "Common/Config.h"
#include "Common/FDUtils.h"
#include "FEXCore/Utils/Allocator.h"
@@ -17,16 +18,15 @@
#include <cstring>
#include <random>
#include <FEXCore/Core/CodeLoader.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/list.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <FEXHeaderUtils/SymlinkChecks.h>
#include <elf.h>
@@ -41,7 +41,7 @@
#define PAGE_OFFSET(x) ((x) & 4095)
#define PAGE_ALIGN(x) (((x) + 4095) & ~(uintptr_t)(4095))
class ELFCodeLoader final : public FEXCore::CodeLoader {
class ELFCodeLoader final : public FEX::CodeLoader {
ELFParser MainElf;
ELFParser InterpElf;
@@ -514,12 +514,12 @@ class ELFCodeLoader final : public FEXCore::CodeLoader {
ASLR_Offset &= (1ULL << ASLR_BITS_32) - 1;
}
ASLR_Offset <<= FHU::FEX_PAGE_SHIFT;
ASLR_Offset <<= FEXCore::Utils::FEX_PAGE_SHIFT;
ELFLoadHint += ASLR_Offset;
}
#endif
// Align the mapping
ELFLoadHint &= FHU::FEX_PAGE_MASK;
ELFLoadHint &= FEXCore::Utils::FEX_PAGE_MASK;
}
// load the main elf
@@ -584,13 +584,13 @@ class ELFCodeLoader final : public FEXCore::CodeLoader {
if (!VSyscallEntry) [[unlikely]] {
// If the VDSO thunk doesn't exist then we might not have a vsyscall entry.
// Newer glibc requires vsyscall to exist now. So let's allocate a buffer and stick a vsyscall in to it.
auto VSyscallPage = Handler->GuestMmap(nullptr, nullptr, FHU::FEX_PAGE_SIZE, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
auto VSyscallPage = Handler->GuestMmap(nullptr, nullptr, FEXCore::Utils::FEX_PAGE_SIZE, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
constexpr static uint8_t VSyscallCode[] = {
0xcd, 0x80, // int 0x80
0xc3, // ret
};
memcpy(VSyscallPage, VSyscallCode, sizeof(VSyscallCode));
mprotect(VSyscallPage, FHU::FEX_PAGE_SIZE, PROT_READ);
mprotect(VSyscallPage, FEXCore::Utils::FEX_PAGE_SIZE, PROT_READ);
VSyscallEntry = reinterpret_cast<uint64_t>(VSyscallPage);
}
+17 -2
View File
@@ -62,7 +62,7 @@ $end_info$
namespace {
static bool SilentLog;
static int OutputFD {STDERR_FILENO};
static int OutputFD {-1};
static bool ExecutedWithFD {false};
void MsgHandler(LogMan::DebugLevels Level, char const *Message) {
@@ -88,7 +88,7 @@ void AssertHandler(char const *Message) {
} // Anonymous namespace
namespace FEXServerLogging {
int FEXServerFD{};
int FEXServerFD{-1};
void MsgHandler(LogMan::DebugLevels Level, char const *Message) {
FEXServerClient::MsgHandler(FEXServerFD, Level, Message);
}
@@ -289,6 +289,7 @@ int main(int argc, char **argv, char **const envp) {
// Early check for process stall
// Doesn't use CONFIG_ROOTFS and we don't want it to spin up a squashfs instance
FEX_CONFIG_OPT(StallProcess, STALLPROCESS);
FEX_CONFIG_OPT(StartupSleep, STARTUPSLEEP);
if (StallProcess) {
while (1) {
// Stall this process out forever
@@ -346,6 +347,11 @@ int main(int argc, char **argv, char **const envp) {
}
}
if (StartupSleep()) {
LogMan::Msg::IFmt("[{}][{}] Sleeping for {} seconds", ::getpid(), Program.ProgramName, StartupSleep());
std::this_thread::sleep_for(std::chrono::seconds(StartupSleep()));
}
FEXCore::Profiler::Init();
FEXCore::Telemetry::Initialize();
@@ -456,6 +462,15 @@ int main(int argc, char **argv, char **const envp) {
auto SyscallHandler = Loader.Is64BitMode() ? FEX::HLE::x64::CreateHandler(CTX.get(), SignalDelegation.get())
: FEX::HLE::x32::CreateHandler(CTX.get(), SignalDelegation.get(), std::move(Allocator));
// Now that we have the syscall handler. Track some FDs that are FEX owned.
if (OutputFD != -1) {
SyscallHandler->FM.TrackFEXFD(OutputFD);
}
SyscallHandler->FM.TrackFEXFD(FEXServerClient::GetServerFD());
if (FEXServerLogging::FEXServerFD != -1) {
SyscallHandler->FM.TrackFEXFD(FEXServerLogging::FEXServerFD);
}
{
// Load VDSO in to memory prior to mapping our ELFs.
void* VDSOBase = FEX::VDSO::LoadVDSOThunks(Loader.Is64BitMode(), SyscallHandler.get());
@@ -205,10 +205,10 @@ enum Syscalls_Arm64 {
SYSCALL_Arm64_semctl = 191,
SYSCALL_Arm64_semtimedop = 192,
SYSCALL_Arm64_semop = 193,
SYSCALL_Arm64__shmget = 194,
SYSCALL_Arm64__shmctl = 195,
SYSCALL_Arm64__shmat = 196,
SYSCALL_Arm64__shmdt = 197,
SYSCALL_Arm64_shmget = 194,
SYSCALL_Arm64_shmctl = 195,
SYSCALL_Arm64_shmat = 196,
SYSCALL_Arm64_shmdt = 197,
SYSCALL_Arm64_socket = 198,
SYSCALL_Arm64_socketpair = 199,
SYSCALL_Arm64_bind = 200,
@@ -339,6 +339,15 @@ enum Syscalls_Arm64 {
SYSCALL_Arm64_set_mempolicy_home_node = 450,
SYSCALL_Arm64_cachestat = 451,
SYSCALL_Arm64_fchmodat2 = 452,
SYSCALL_Arm64_map_shadow_stack = 453,
SYSCALL_Arm64_futex_wake = 454,
SYSCALL_Arm64_futex_wait = 455,
SYSCALL_Arm64_futex_requeue = 456,
SYSCALL_Arm64_statmount = 457,
SYSCALL_Arm64_listmount = 458,
SYSCALL_Arm64_lsm_get_self_attr = 459,
SYSCALL_Arm64_lsm_set_self_attr = 460,
SYSCALL_Arm64_lsm_list_modules = 461,
SYSCALL_Arm64_MAX = 512,
// Unsupported syscalls on this host
@@ -466,6 +475,5 @@ enum Syscalls_Arm64 {
SYSCALL_Arm64_epoll_ctl_old = ~0,
SYSCALL_Arm64_epoll_wait_old = ~0,
SYSCALL_Arm64_newfstatat = ~0,
SYSCALL_Arm64_map_shadow_stack = ~0,
};
}
@@ -6,12 +6,13 @@ desc: Emulated /proc/cpuinfo, version, osrelease, etc
$end_info$
*/
#include "CodeLoader.h"
#include "Common/FDUtils.h"
#include "LinuxSyscalls/Syscalls.h"
#include "LinuxSyscalls/EmulatedFiles/EmulatedFiles.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CodeLoader.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Utils/CPUInfo.h>
@@ -219,6 +219,7 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
// - AppConfig Global
// - Steam AppConfig Local
// - AppConfig Local
// - AppConfig override
// This doesn't support the classic thunks interface.
auto AppName = AppConfigName();
@@ -245,11 +246,19 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
ConfigPaths.emplace_back(FEXCore::Config::GetApplicationConfig(AppName, false));
}
const char *AppConfig = getenv("FEX_APP_CONFIG");
if (AppConfig) {
ConfigPaths.emplace_back(AppConfig);
}
if (!LDPath().empty()) {
RootFSFD = open(LDPath().c_str(), O_DIRECTORY | O_PATH | O_CLOEXEC);
if (RootFSFD == -1) {
RootFSFD = AT_FDCWD;
}
else {
TrackFEXFD(RootFSFD);
}
}
fextl::unordered_map<fextl::string, ThunkDBObject> ThunkDB;
@@ -571,6 +580,13 @@ uint64_t FileManager::Open(const char *pathname, int flags, uint32_t mode) {
}
uint64_t FileManager::Close(int fd) {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
if (CheckIfFDInTrackedSet(fd)) {
LogMan::Msg::EFmt("{} closing FEX FD {}", __func__, fd);
RemoveFEXFD(fd);
}
#endif
return ::close(fd);
}
@@ -578,6 +594,14 @@ uint64_t FileManager::CloseRange(unsigned int first, unsigned int last, unsigned
#ifndef CLOSE_RANGE_CLOEXEC
#define CLOSE_RANGE_CLOEXEC (1U << 2)
#endif
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
if (!(flags & CLOSE_RANGE_CLOEXEC) &&
CheckIfFDRangeInTrackedSet(first, last)) {
LogMan::Msg::EFmt("{} closing FEX FDs in range ({}, {})", __func__, first, last);
RemoveFEXFDRange(first, last);
}
#endif
return ::syscall(SYSCALL_DEF(close_range), first, last, flags);
}
@@ -88,7 +88,53 @@ public:
bool SupportsProcFSInterpreterPath() const { return SupportsProcFSInterpreter; }
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
void TrackFEXFD(int FD) noexcept {
std::lock_guard lk(FEXTrackingFDMutex);
FEXTrackingFDs.emplace(FD);
}
void RemoveFEXFD(int FD) noexcept {
std::lock_guard lk(FEXTrackingFDMutex);
FEXTrackingFDs.erase(FD);
}
void RemoveFEXFDRange(int begin, int end) noexcept {
std::lock_guard lk(FEXTrackingFDMutex);
std::erase_if(FEXTrackingFDs, [begin, end](int FD) {
return FD >= begin && (FD <= end || end == -1);
});
}
bool CheckIfFDInTrackedSet(int FD) noexcept {
std::lock_guard lk(FEXTrackingFDMutex);
return FEXTrackingFDs.contains(FD);
}
bool CheckIfFDRangeInTrackedSet(int begin, int end) noexcept {
std::lock_guard lk(FEXTrackingFDMutex);
// Just linear scan since the number of tracking FDs is low.
for (auto it : FEXTrackingFDs) {
if (it >= begin && (it <= end || end == -1)) return true;
}
return false;
}
#else
void TrackFEXFD(int FD) const noexcept {}
bool CheckIfFDInTrackedSet(int FD) const noexcept { return false; }
void RemoveFEXFD(int FD) const noexcept {}
void RemoveFEXFDRange(int begin, int end) const noexcept {}
bool CheckIfFDRangeInTrackedSet(int begin, int end) const noexcept { return false; }
#endif
private:
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
std::mutex FEXTrackingFDMutex;
fextl::set<int> FEXTrackingFDs;
#endif
bool RootFSPathExists(const char* Filepath);
struct ThunkDBObject {
Loaded 100 of 208 files, more files were not shown because too many files have changed in this diff. Show more