mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 17:00:19 +02:00
JIT: Optimize x87 FSINCOS
Turns out Bayonetta hammers SINCOS, our splitting the operation is actually harming the performance of games that heavily use FSINCOS. We instead can actually combine the operation which improves performance. Not enough to get the game running full speed consistently on my Radxa, but good numbers in my microbenchmark. ``` Test, Total Cycles, Total Runs, Cycles Average, Internal Loops, Average cycles per internal, per/second 64-bit: Before: FSIN, 2691031290, 50000, 53820.6, 1000, 53.8206, 18580.237319 FCOS, 2719397120, 50000, 54387.9, 1000, 54.3879, 18386.428239 FSINCOS, 5586917530, 50000, 111738, 1000, 111.738, 8949.478801 After: FSIN, 2669959250, 50000, 53399.2, 1000, 53.3992, 18726.877573 FCOS, 2740942260, 50000, 54818.8, 1000, 54.8188, 18241.901965 FSINCOS, 3189472870, 50000, 63789.5, 1000, 63.7895, 15676.571659 80-bit: Before: FSIN, 24702939380, 50000, 494059, 1000, 494.059, 2024.050629 FCOS, 19127131020, 50000, 382543, 1000, 382.543, 2614.087808 FSINCOS, 40386785260, 50000, 807736, 1000, 807.736, 1238.028719 After: FSIN, 24869980710, 50000, 497400, 1000, 497.4, 2010.455922 FCOS, 19131849590, 50000, 382637, 1000, 382.637, 2613.443084 FSINCOS, 38329985570, 50000, 766600, 1000, 766.6, 1304.461749 Improvement 64-bit: 1.75x Improvement 80-bit: 1.05x ``` Only a minor improvement at 80-bit precision since cephes doesn't provide a combined sincos operation, but the f64 implementation is significantly improved, allowing 75% more operations per second. Disabled in the simulator because we can't easily support pairs of vector registers being returned.
This commit is contained in:
9 files changed
+218
-6
No files matched your search
@@ -161,6 +161,7 @@ private:
|
||||
const FEXCore::HostFeatures& Features;
|
||||
const OpSize GPROpSize;
|
||||
bool ReducedPrecisionMode;
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
|
||||
// Helpers
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
@@ -774,12 +775,26 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref SinValue {};
|
||||
Ref CosValue {};
|
||||
if (ReducedPrecisionMode) {
|
||||
SinValue = IREmit->_F64SIN(St0);
|
||||
CosValue = IREmit->_F64COS(St0);
|
||||
} else {
|
||||
SinValue = IREmit->_F80SIN(St0);
|
||||
CosValue = IREmit->_F80COS(St0);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
if (DisableVixlIndirectCalls() == 0) {
|
||||
if (ReducedPrecisionMode) {
|
||||
SinValue = IREmit->_F64SIN(St0);
|
||||
CosValue = IREmit->_F64COS(St0);
|
||||
} else {
|
||||
SinValue = IREmit->_F80SIN(St0);
|
||||
CosValue = IREmit->_F80COS(St0);
|
||||
}
|
||||
} else
|
||||
#endif
|
||||
{
|
||||
SinValue = IREmit->_AllocateFPR(OpSize::i128Bit, OpSize::i128Bit);
|
||||
CosValue = IREmit->_AllocateFPR(OpSize::i128Bit, OpSize::i128Bit);
|
||||
if (ReducedPrecisionMode) {
|
||||
IREmit->_F64SINCOS(St0, SinValue, CosValue);
|
||||
} else {
|
||||
IREmit->_F80SINCOS(St0, SinValue, CosValue);
|
||||
}
|
||||
}
|
||||
|
||||
// Push values
|
||||
|
||||
Reference in new issue
Block a user