Refactor stereo frame worker and interpolation tests for enhanced VR performance

- Updated stereo_frame_worker_smoke.cpp to allow dynamic headset rates and prediction lead time.
- Improved logging to include motion diagnostics and adjusted frame submission logic based on headset frequency.
- Enhanced stereo_interpolation_test.cpp with additional tests for camera motion separation and playback cadence.
- Introduced MkwVRReadSceneView function to read the camera view matrix for improved scene rendering.
- Modified VR first-person logic to support scene view reading and validation.
- Added scene_camera.hpp to encapsulate camera motion handling and inverse view calculations.
- Ensured that the VR integration layer correctly logs motion diagnostics and handles scene playback accurately.
This commit is contained in:
iChris4 committed 2026-10-01 00:37:49 +02:00
1 parent 85fab2fa09
commit e7eab8a6b2
19 files changed
+1978 -114

No files matched your search

+6 -1
View File
@@ -197,7 +197,8 @@ typedef struct {
AuroraStereoFrameMode mode;
uint64_t contentTag;
// Predicted display time converted to std::chrono::steady_clock nanoseconds.
// Zero disables temporal interpolation for this packet.
// Used for diagnostics. Scene playback has its own clock; the eye poses remain
// predicted for this time even when the runtime looks several game frames ahead.
uint64_t displayTimeNanos;
// Optional; inactive when zero-initialised.
AuroraCockpit cockpit;
@@ -334,6 +335,10 @@ void aurora_set_stereo_scene_anchor(const float anchorFromScene[12]);
// The sealed frame then owns that scale: each eye's head/IPD translation is
// rescaled from the packet's AuroraCockpit::unitsPerMeter to it.
void aurora_set_stereo_scene_anchor_scaled(const float anchorFromScene[12], float unitsPerMeter);
// Recorded world-to-view camera, copied on the GX thread beside the scene anchor.
// Null disables camera-separated interpolation for this frame.
void aurora_set_stereo_scene_view(const float viewFromWorld[12]);
// GX producer thread, after the scene anchor and before sealing that frame.
void aurora_set_stereo_cockpit_item(const AuroraCockpitItem* item);
// Copies a user-supplied Race/Common.szs archive. May be called on the guest
+8
View File
@@ -63,6 +63,12 @@ typedef struct {
uint64_t framesReplayUnsafe;
uint64_t slotReductions;
uint64_t lateSealDrops;
// Matched identity does not guarantee usable interpolation transforms.
uint32_t preparedDraws; // latest seal: draws with usable endpoint pairs
uint32_t rejectedDraws; // latest seal: paired draws rejected by transform guards
uint32_t vertexMotionDraws; // latest seal: CPU-authored quads with usable motion pairs
uint32_t vertexMotionHeld; // latest seal: such quads drawn where recorded (no unambiguous partner yet)
uint64_t animationWrapCuts; // cumulative rigid animation resets held at their new phase
} AuroraFrameInterpolationDiagnostics;
void aurora_get_frame_interpolation_diagnostics(AuroraFrameInterpolationDiagnostics* diagnostics);
@@ -76,6 +82,8 @@ uint32_t aurora_get_frame_interpolation_fps();
// each headset deadline, leaving guest simulation and VI timing at 60 Hz.
void aurora_set_stereo_frame_interpolation(bool enabled);
bool aurora_get_stereo_frame_interpolation();
// Logs scene sampling and transform acceptance separately from headset submission FPS.
void aurora_set_stereo_motion_logging(bool enabled);
// Newly encountered GX pipelines compile on the bounded worker queue. Draws whose pipeline is not
// ready are skipped rather than stalling submission, and pick it up once compilation finishes.
+112 -8
View File
@@ -11,6 +11,7 @@
#include "stereo.hpp"
#include "stereo_mirror.hpp"
#include "stereo_interpolation.hpp"
#include "scene_camera.hpp"
#include "stereo_overlay.hpp"
#include "webgpu/fdm.hpp"
#include "webgpu/gpu.hpp"
@@ -79,6 +80,7 @@ std::atomic<AuroraFrameLogCallback> g_frameLogCallback{nullptr};
// deadlines derived from it, so the presenter cannot drift. Zero means present when ready.
std::atomic<uint64_t> g_presentScheduleBaseNanos{0};
std::atomic<uint64_t> g_presentScheduleIntervalNanos{0};
std::atomic_bool g_stereoMotionLogging{false};
namespace {
Module Log("aurora");
@@ -119,6 +121,8 @@ struct StereoSceneAnchor {
// own scale applies (aurora_set_stereo_scene_anchor_scaled).
float unitsPerMeter = 0.f;
AuroraCockpitItem cockpitItem{};
std::array<float, 12> viewFromWorld{};
bool viewValid = false;
};
// Producer thread only, between aurora_set_stereo_scene_anchor() and the seal
// that consumes it. Cleared at every seal so a producer that stops publishing
@@ -1940,15 +1944,78 @@ struct RetainedStereoContext {
StereoSceneAnchor anchor;
StereoSceneAnchor previousAnchor;
bool continuous = false;
stereo::SceneCameraMotion cameraMotion;
} g_retainedStereo;
stereo::ScenePlaybackClock g_stereoPlaybackClock;
void log_stereo_motion(const AuroraStereoFrame& input, uint64_t sampleTime, float weight) {
// Environment switch also works for hosts using Aurora without OpenXR.
static const bool forced = [] {
const char* value = std::getenv("AURORA_VR_MOTION_LOG");
bool enabled = value != nullptr && std::strcmp(value, "1") == 0;
#if defined(__ANDROID__)
enabled = enabled || android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
#endif
return enabled;
}();
static stereo::MotionSamples samples;
static uint32_t maxRejected = 0;
static auto start = std::chrono::steady_clock::now();
if (!forced && !g_stereoMotionLogging.load(std::memory_order_relaxed)) {
samples = {};
maxRejected = 0;
start = std::chrono::steady_clock::now();
return;
}
const auto& retained = g_retainedStereo;
AuroraFrameInterpolationDiagnostics draws{};
gx::get_frame_interpolation_diagnostics(draws);
maxRejected = std::max(maxRejected, draws.rejectedDraws);
if (samples.samples == 0 && samples.lastSceneTime == 0)
start = std::chrono::steady_clock::now();
samples.record(sampleTime, retained.boundary, retained.interval, retained.continuous, weight);
const auto now = std::chrono::steady_clock::now();
const double seconds = std::chrono::duration<double>(now - start).count();
if (seconds < 1.0)
return;
const auto deltaMs = [](uint64_t a, uint64_t b) {
return a >= b ? static_cast<double>(a - b) / 1e6 : -static_cast<double>(b - a) / 1e6;
};
const uint64_t nowNs = std::chrono::duration_cast<std::chrono::nanoseconds>(now.time_since_epoch()).count();
Log.info("[vr-motion] {:.2f}s samples={} blended={} previous={} current={} discontinuous={} "
"same-time={} backwards={} scene-step={:.2f}/{:.2f}ms max-rejected={} | latest "
"display-boundary={:.2f}ms prediction-lead={:.2f}ms sample-boundary={:.2f}ms weight={:.3f} | latest draws "
"candidates={} matched={} prepared={} rejected={} replay-safe={} camera-separated={} "
"vertex-motion={} vertex-held={} wrap-cuts={}",
seconds, samples.samples, samples.blended, samples.atPrevious, samples.atCurrent,
samples.discontinuous, samples.repeated, samples.backwards,
samples.minStep == UINT64_MAX ? 0.0 : samples.minStep / 1e6, samples.maxStep / 1e6, maxRejected,
deltaMs(input.displayTimeNanos, retained.boundary), deltaMs(input.displayTimeNanos, nowNs),
deltaMs(sampleTime, retained.boundary), weight,
draws.candidates, draws.matches, draws.preparedDraws, draws.rejectedDraws, draws.replaySafe,
retained.cameraMotion.active, draws.vertexMotionDraws, draws.vertexMotionHeld, draws.animationWrapCuts);
samples.clear_window();
maxRejected = 0;
start = now;
}
gfx::StereoReplayFrame interpolated_stereo_frame(const AuroraStereoFrame& input, float& weight) {
const auto& retained = g_retainedStereo;
const uint64_t now = std::chrono::duration_cast<std::chrono::nanoseconds>(
std::chrono::steady_clock::now().time_since_epoch()).count();
const uint64_t sampleTime = g_stereoPlaybackClock.sample_time(now);
weight = retained.continuous
? stereo::interpolation_weight(input.displayTimeNanos, retained.boundary, retained.interval)
? stereo::interpolation_weight(sampleTime, retained.boundary, retained.interval)
: 1.0f;
log_stereo_motion(input, sampleTime, weight);
auto anchor = retained.anchor;
if (weight < 1.0f && anchor.active && retained.previousAnchor.active) {
if (weight < 1.0f && retained.cameraMotion.active) {
Mat3x4<float> result;
if (retained.cameraMotion.sample(weight, result)) {
std::memcpy(anchor.anchorFromScene.data(), &result, sizeof(result));
anchor.active = true;
}
} else if (weight < 1.0f && anchor.active && retained.previousAnchor.active) {
Mat3x4<float> previous, current, result;
std::memcpy(&previous, retained.previousAnchor.anchorFromScene.data(), sizeof(previous));
std::memcpy(&current, anchor.anchorFromScene.data(), sizeof(current));
@@ -2004,6 +2071,27 @@ void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, u
// pre-first-frame UINT32_MAX value to logical frame zero.
ctx.logicalFrame = gfx::current_frame() + 1;
gfx::set_stereo_local_player_count(sceneAnchor.localPlayerCount);
ctx.scheduleBaseNanos = g_presentScheduleBaseNanos.load(std::memory_order_acquire);
ctx.scheduleIntervalNanos = g_presentScheduleIntervalNanos.load(std::memory_order_acquire);
bool continuous = g_retainedStereo.contentTag == contentTag &&
g_retainedStereo.interval != 0 && ctx.scheduleIntervalNanos != 0 &&
ctx.scheduleBaseNanos > g_retainedStereo.boundary &&
ctx.scheduleBaseNanos - g_retainedStereo.boundary <= ctx.scheduleIntervalNanos * 3 / 2 &&
g_retainedStereo.anchor.active == sceneAnchor.active &&
g_retainedStereo.anchor.localPlayerCount == sceneAnchor.localPlayerCount;
stereo::SceneCameraMotion cameraMotion;
if (continuous && gx::stereo_frame_interpolation_active() && sceneAnchor.localPlayerCount == 1 &&
sceneAnchor.viewValid && g_retainedStereo.anchor.viewValid) {
Mat3x4<float> previousView, currentView, previousAnchor, currentAnchor;
std::memcpy(static_cast<void*>(&previousView), g_retainedStereo.anchor.viewFromWorld.data(), sizeof(previousView));
std::memcpy(static_cast<void*>(&currentView), sceneAnchor.viewFromWorld.data(), sizeof(currentView));
std::memcpy(static_cast<void*>(&previousAnchor), g_retainedStereo.anchor.anchorFromScene.data(), sizeof(previousAnchor));
std::memcpy(static_cast<void*>(&currentAnchor), sceneAnchor.anchorFromScene.data(), sizeof(currentAnchor));
// A real camera cut uses current endpoints for the whole scene.
continuous = cameraMotion.prepare(previousView, currentView, previousAnchor, currentAnchor);
}
gx::set_frame_interpolation_view_rebase(cameraMotion.active ? &cameraMotion.currentFromPrevious : nullptr,
cameraMotion.active ? &cameraMotion.previousFromCurrent : nullptr);
if (const auto stereoInput = request_stereo_frame(ctx.logicalFrame, contentTag)) {
ctx.stereoInput = stereoInput;
ctx.stereoFrameToken = stereoInput->frameToken;
@@ -2061,15 +2149,14 @@ void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, u
// encode phase reads only worker-private state.
gfx::seal_frame(sealedFrame);
ctx.retainStereo = gx::stereo_frame_interpolation_active() && gfx::has_late_stereo_replay(sealedFrame);
const bool continuous = ctx.retainStereo && g_retainedStereo.contentTag == contentTag &&
g_retainedStereo.interval != 0 && ctx.scheduleIntervalNanos != 0 &&
ctx.scheduleBaseNanos > g_retainedStereo.boundary &&
ctx.scheduleBaseNanos - g_retainedStereo.boundary <= ctx.scheduleIntervalNanos * 3 / 2 &&
g_retainedStereo.anchor.active == sceneAnchor.active;
continuous = continuous && ctx.retainStereo;
const auto previousAnchor = g_retainedStereo.anchor;
g_retainedStereo = {contentTag, ctx.scheduleBaseNanos, ctx.scheduleIntervalNanos,
ctx.logicalFrame, sceneAnchor, previousAnchor,
continuous};
continuous, cameraMotion};
const uint64_t now = std::chrono::duration_cast<std::chrono::nanoseconds>(
std::chrono::steady_clock::now().time_since_epoch()).count();
g_stereoPlaybackClock.begin_scene(ctx.scheduleBaseNanos, now, continuous);
gfx::expire_bind_group_cache();
}
@@ -2770,6 +2857,19 @@ void aurora_set_stereo_scene_anchor_scaled(const float anchorFromScene[12], floa
}
}
void aurora_set_stereo_scene_view(const float viewFromWorld[12]) {
auto& pending = aurora::g_pendingSceneAnchor;
pending.viewValid = false;
if (viewFromWorld == nullptr) return;
for (size_t i = 0; i < 12; ++i) {
uint32_t bits;
std::memcpy(&bits, viewFromWorld + i, sizeof(bits));
if ((bits & 0x7f800000u) == 0x7f800000u) return;
}
std::memcpy(pending.viewFromWorld.data(), viewFromWorld, sizeof(pending.viewFromWorld));
pending.viewValid = true;
}
void aurora_set_stereo_cockpit_item(const AuroraCockpitItem* item) {
aurora::set_stereo_cockpit_item(item);
}
@@ -2848,6 +2948,10 @@ void aurora_get_frame_interpolation_diagnostics(AuroraFrameInterpolationDiagnost
}
}
void aurora_set_stereo_motion_logging(bool enabled) {
aurora::g_stereoMotionLogging.store(enabled, std::memory_order_relaxed);
}
void aurora_get_present_timing(AuroraPresentTiming* timing) {
if (timing != nullptr) {
*timing = aurora::snapshot_present_timing();
+11 -2
View File
@@ -2188,10 +2188,19 @@ bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, c
Mat3x4<float> before, current, result;
std::memcpy(&before, previous + at, sizeof(before));
std::memcpy(&current, uniform.data() + at, sizeof(current));
const bool vertexMotion = layout.vertexMotion.enabled && offset == layout.positionOffset &&
!layout.indexedMatrices;
if (vertexMotion) {
before = gx::offset_transform_origin(before, layout.vertexMotion.center);
current = gx::offset_transform_origin(current, layout.vertexMotion.center);
}
const bool valid = layout.indexedMatrices ? gx::interpolate_indexed_transform(before, current, weight, result)
: gx::interpolate_transform(before, current, weight, result);
if (valid)
: gx::interpolate_draw_transform(before, current, weight, result);
if (valid) {
if (vertexMotion)
result = gx::offset_transform_origin(result, layout.vertexMotion.center, -1.f);
std::memcpy(uniform.data() + at, &result, sizeof(result));
}
}
};
interpolateMatrices(layout.positionOffset, layout.positionMatrixCount, layout.positionMatrixMask);
+41 -3
View File
@@ -2379,6 +2379,17 @@ bool submit_raw_draw(GXPrimitive prim, GXVtxFmt fmt, const uint8_t* vertices, ui
return true;
}
// Runs for every draw, VR interpolation or not: the shape tests come first so that
// ordinary draws never reach the atomic flag.
static bool particle_quad_motion(GXPrimitive prim, GXVtxFmt fmt, uint16_t count) noexcept {
if (prim != GX_QUADS || count != 4 || g_gxState.vtxDesc[GX_VA_POS] != GX_DIRECT) {
return false;
}
const auto& attr = g_gxState.vtxFmts[fmt].attrs[GX_VA_POS];
return g_gxState.vtxDesc[GX_VA_PNMTXIDX] == GX_NONE && attr.cnt == GX_POS_XYZ && attr.type == GX_F32 &&
g_gxState.projType == GX_PERSPECTIVE && stereo_frame_interpolation_active();
}
static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndian) {
ZoneScoped;
GXVtxFmt fmt = static_cast<GXVtxFmt>(cmd & CP_VAT_MASK);
@@ -2418,13 +2429,15 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi
NativeWheelArray* const nativeWheel = resolve_native_wheel(fmt, vertices, vtxCount, vtxSize);
// Try to merge with previous draw call.
if (!g_gxState.stateDirty && !(aurora::stereo_frame_provider_active() && g_gxState.projType == GX_ORTHOGRAPHIC))
if (!g_gxState.stateDirty && !particle_quad_motion(prim, fmt, vtxCount) &&
!(aurora::stereo_frame_provider_active() && g_gxState.projType == GX_ORTHOGRAPHIC))
LIKELY {
auto* lastDraw = gfx::get_last_draw_command<DrawData>();
// Only if the previous draw call was a single instance draw (no lines/points handling), and only into a draw
// that resolved the same animated array: the merged whole renders through that draw's binding. Anything the
// decision cache cannot vouch for (a command it was not recorded against) stays unmerged.
if (lastDraw != nullptr && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS &&
!lastDraw->uniformReplayLayout.vertexMotion.enabled &&
lastDraw->instanceCount == 1 &&
(nativeWheelArrays.empty() ||
(nativeWheelLastDrawCommand == lastDraw && nativeWheelLastDecision == nativeWheel)))
@@ -2527,11 +2540,34 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
// Draw-identity hashing only feeds frame interpolation, and only for perspective draws: build_uniform reads the
// identity exclusively past its `!perspective || frame_interpolation_fps() == 0` early-out.
FrameInterpolationDrawIdentity drawIdentity{};
DrawVertexMotion vertexMotion{};
DrawVertexShape vertexShape{};
if (interpolationIdentityActive && particle_quad_motion(prim, fmt, vtxCount)) {
const uint32_t offset = matrix_index_prefix_size(fmt);
if (offset + 3 * sizeof(float) <= vtxStride) {
vertexMotion.enabled = true;
std::array<std::array<float, 3>, 4> corners{};
for (uint32_t vertex = 0; vertex < 4; ++vertex) {
for (uint32_t component = 0; component < 3; ++component) {
const auto* at = vertices + vertex * vtxStride + offset + component * sizeof(float);
// Bitwise finite test also works in fast-math product builds.
vertexMotion.enabled &= (read_u32(at, true) & 0x7f800000u) != 0x7f800000u;
corners[vertex][component] = read_f32(at, true);
vertexMotion.center[component] += corners[vertex][component] * 0.25f;
}
}
for (uint32_t component = 0; component < 3; ++component) {
vertexShape.edge0[component] = corners[1][component] - corners[0][component];
vertexShape.edge1[component] = corners[3][component] - corners[0][component];
}
}
}
if (interpolationIdentityActive)
UNLIKELY {
const HashType drawShape = static_cast<HashType>(vtxCount) | (static_cast<HashType>(underlying(prim)) << 16) |
(static_cast<HashType>(underlying(fmt)) << 24);
const HashType pipelineDrawSignature = xxh3_hash(pipelineState.configHash, drawShape);
const HashType pipelineDrawSignature =
xxh3_hash(pipelineState.configHash, drawShape | (HashType(vertexMotion.enabled) << 32));
const HashType textureSignature = xxh3_hash(bindGroups.textureBindGroup);
const HashType materialAndTopology =
xxh3_hash(matrixTopologySignature, xxh3_hash(bindGroups.textureBindGroup, pipelineDrawSignature));
@@ -2540,10 +2576,12 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
.pipeline = pipelineDrawSignature,
.texture = textureSignature,
.matrixTopology = matrixTopologySignature,
.geometry = geometrySignature,
};
}
const bool perspective = g_gxState.projType == GX_PERSPECTIVE;
const auto uniformRanges = build_uniform(info, vertRange.offset, ranges, drawIdentity, perspective, usedPnMtxMask);
const auto uniformRanges = build_uniform(info, vertRange.offset, ranges, drawIdentity, perspective, usedPnMtxMask,
vertexMotion, vertexShape);
const auto& replayLayout = uniformRanges.replayLayout;
const bool stereo = aurora::stereo_frame_provider_active();
const bool screen = !replayLayout.perspective && !replayLayout.nativeEfbEffect;
+546 -72
View File
@@ -2,6 +2,7 @@
#include "../internal.hpp"
#include "aurora/gfx.h"
#include "../gfx/stereo_replay.hpp"
// Guest matrices really do carry NaN/Inf, and the isfinite guards here keep them
// out of the MatchEdge sort. Needs -fno-finite-math-only (see runtime/CMakeLists.txt).
@@ -44,6 +45,11 @@ std::atomic_uint64_t s_diagFramesLowMatch{0};
std::atomic_uint64_t s_diagFramesReplayUnsafe{0};
std::atomic_uint64_t s_diagSlotReductions{0};
std::atomic_uint64_t s_diagLateSealDrops{0};
std::atomic_uint32_t s_diagPreparedDraws{0};
std::atomic_uint32_t s_diagRejectedDraws{0};
std::atomic_uint32_t s_diagVertexMotionDraws{0};
std::atomic_uint32_t s_diagVertexMotionHeld{0};
std::atomic_uint64_t s_diagAnimationWrapCuts{0};
// Persistent worker pool for the per-sample interpolation tasks. libc++ has no
// parallel execution policies, so without it the seal loop runs serially. Leaked.
class InterpolationWorkerPool {
@@ -147,6 +153,19 @@ struct FrameTransformSnapshot {
Mat3x4<float> position{};
Mat3x4<float> normal{};
uint16_t usedMatrixMask = 1;
// Direct particle vertices can already be in camera space with identity XF.
// Their retained vertex buffer belongs to the current frame, so only the
// sampled camera should move them; an old matrix cannot animate those vertices.
bool viewSpaceVertices = false;
DrawVertexMotion vertexMotion{};
// Vertex-motion quads only: centre and edges in the recording camera's space.
// The seal rebases `position` as if the quad were world-fixed, but emitters
// that follow the kart keep their particles near their old camera-space place.
std::array<float, 3> quadCenter{};
std::array<std::array<float, 3>, 2> quadEdges{};
// Left unmatched in a group whose particles follow the camera: hold it there,
// not at a world position the sampled camera would sweep past.
bool holdInCamera = false;
struct IndexedMatrices {
std::array<Mat3x4<float>, MaxPnMtx> position{};
std::array<Mat3x4<float>, MaxPnMtx> normal{};
@@ -189,10 +208,14 @@ struct PendingUniformInterpolation {
std::vector<FrameTransformEntry> s_previousFrameTransforms;
std::vector<FrameTransformEntry> s_currentFrameTransforms;
Mat3x4<float> s_currentFromPreviousView{}, s_previousFromCurrentView{};
bool s_rebaseView = false;
std::unordered_map<HashType, std::vector<size_t>> s_previousTransformIndices;
std::unordered_map<HashType, std::vector<size_t>> s_currentTransformIndices;
std::unordered_map<HashType, std::vector<size_t>> s_previousStableTransformIndices;
std::unordered_map<HashType, std::vector<size_t>> s_currentStableTransformIndices;
std::unordered_map<HashType, std::vector<size_t>> s_previousGeometryTransformIndices;
std::unordered_map<HashType, std::vector<size_t>> s_currentGeometryTransformIndices;
// Free list for the indexed-matrix snapshots; per-draw heap allocation was the
// hottest cost in this path. Unused slots keep stale data, consumers mask first.
@@ -285,6 +308,19 @@ constexpr float kMinimumScale = 1.0e-5f;
constexpr float kMaximumTranslationPerFrame = 1500.0f;
constexpr float kMinimumQuaternionDot = 0.70710678f;
constexpr size_t kNoPreparedPair = std::numeric_limits<size_t>::max();
// Vertex-motion quad pairing (matchQuadGroup). Costs are squared distances.
// A pair must undercut the runner-up among other particles by this factor.
constexpr float kQuadAmbiguity = 1.5f;
// A particle's edges change by at most 30% of its size per frame.
constexpr float kQuadShapeChange = 0.3f * 0.3f;
// A particle with a path lands within a quarter of its last step of the
// prediction, or within a tenth of its short edge when it barely moves.
constexpr float kQuadPathTolerance = 0.25f * 0.25f;
constexpr float kQuadPathFloor = 0.1f * 0.1f;
// Centres this close belong to one particle: a cross draws two quads per streak.
constexpr float kQuadSamePlace = 1.0e-4f;
// All-pairs bound per group. A denser swarm stays where the game drew it.
constexpr size_t kMaximumQuadPairs = 4096;
HashType combine_identity(HashType first, HashType second) noexcept {
return xxh3_hash(second, first);
@@ -295,6 +331,10 @@ HashType stable_identity(const FrameInterpolationDrawIdentity& identity) noexcep
identity.matrixTopology);
}
HashType geometry_identity(const FrameInterpolationDrawIdentity& identity) noexcept {
return combine_identity(combine_identity(identity.pipeline, identity.geometry), identity.matrixTopology);
}
float dot3(const std::array<float, 3>& a, const std::array<float, 3>& b) noexcept {
return a[0] * b[0] + a[1] * b[1] + a[2] * b[2];
}
@@ -338,50 +378,53 @@ Quaternion quaternion_from_rotation(const std::array<std::array<float, 3>, 3>& m
return q;
}
// Splits the 3x3 part into a rotation R and an upper-triangular stretch U with M = R*U
// (Gram-Schmidt on the columns): scale on U's diagonal, shear above it, a mirror as a
// negative last diagonal. V*T*R*S puts non-uniform model scale on the columns, so a plain
// R*S comes back with a diagonal U. A sheared matrix keeps its tilt in U instead of
// losing it to the nearest rotation. `skew` receives the largest |cosine| between the
// columns: zero for rotation times scale.
bool decompose_affine(const Mat3x4<float>& matrix, std::array<std::array<float, 3>, 3>& rotation,
std::array<float, 3>& scale, std::array<float, 3>& translation,
Quaternion& quaternion) noexcept {
std::array<float, 6>& stretch, std::array<float, 3>& translation,
Quaternion& quaternion, float& skew) noexcept {
const std::array<Vec4<float>, 3> rows{matrix.m0, matrix.m1, matrix.m2};
// V*T*R*S puts non-uniform model scale on the columns, so scale must be read off the
// columns; row extraction misreads R*S as shear whenever the rotation tilts an axis.
for (size_t row = 0; row < 3; ++row) {
translation[row] = rows[row].w();
if (!std::isfinite(translation[row])) {
return false;
}
}
for (size_t column = 0; column < 3; ++column) {
const float x = rows[0][column];
const float y = rows[1][column];
const float z = rows[2][column];
scale[column] = std::sqrt(x * x + y * y + z * z);
if (!std::isfinite(scale[column]) || scale[column] < kMinimumScale) {
return false;
}
rotation[0][column] = x / scale[column];
rotation[1][column] = y / scale[column];
rotation[2][column] = z / scale[column];
}
const auto rotationColumn = [&rotation](size_t column) noexcept {
return std::array<float, 3>{rotation[0][column], rotation[1][column], rotation[2][column]};
const auto column = [&rows](size_t index) noexcept {
return std::array<float, 3>{rows[0][index], rows[1][index], rows[2][index]};
};
if (std::abs(dot3(rotationColumn(0), rotationColumn(1))) > 0.05f ||
std::abs(dot3(rotationColumn(0), rotationColumn(2))) > 0.05f ||
std::abs(dot3(rotationColumn(1), rotationColumn(2))) > 0.05f) {
const auto c0 = column(0), c1 = column(1), c2 = column(2);
const float u00 = std::sqrt(dot3(c0, c0));
if (!std::isfinite(u00) || u00 < kMinimumScale) {
return false;
}
const float determinant = dot3(rotation[0], cross3(rotation[1], rotation[2]));
if (!std::isfinite(determinant) || std::abs(std::abs(determinant) - 1.0f) > 0.1f) {
const std::array<float, 3> r0{c0[0] / u00, c0[1] / u00, c0[2] / u00};
const float u01 = dot3(r0, c1);
const std::array<float, 3> c1Rest{c1[0] - u01 * r0[0], c1[1] - u01 * r0[1], c1[2] - u01 * r0[2]};
const float u11 = std::sqrt(dot3(c1Rest, c1Rest));
if (!std::isfinite(u11) || u11 < kMinimumScale) {
return false;
}
if (determinant < 0.0f) {
scale[0] = -scale[0];
for (size_t row = 0; row < 3; ++row) {
rotation[row][0] = -rotation[row][0];
}
const std::array<float, 3> r1{c1Rest[0] / u11, c1Rest[1] / u11, c1Rest[2] / u11};
const auto r2 = cross3(r0, r1);
const float u02 = dot3(r0, c2);
const float u12 = dot3(r1, c2);
const float u22 = dot3(r2, c2);
if (!std::isfinite(u02) || !std::isfinite(u12) || !std::isfinite(u22) || std::abs(u22) < kMinimumScale) {
return false;
}
const float length1 = std::sqrt(dot3(c1, c1));
const float length2 = std::sqrt(dot3(c2, c2));
skew = std::max({std::abs(u01) / length1, std::abs(u02) / length2,
std::abs(dot3(c1, c2)) / (length1 * length2)});
for (size_t row = 0; row < 3; ++row) {
rotation[row] = {r0[row], r1[row], r2[row]};
}
stretch = {u00, u01, u02, u11, u12, u22};
quaternion = quaternion_from_rotation(rotation);
return std::isfinite(quaternion.x) && std::isfinite(quaternion.y) && std::isfinite(quaternion.z) &&
@@ -408,12 +451,15 @@ std::array<std::array<float, 3>, 3> rotation_from_quaternion(const Quaternion& q
struct PreparedAffinePair {
Mat3x4<float> previous{};
Mat3x4<float> current{};
std::array<float, 3> previousScale{};
std::array<float, 3> currentScale{};
// Upper-triangular stretch from decompose_affine: u00 u01 u02 u11 u12 u22.
std::array<float, 6> previousStretch{};
std::array<float, 6> currentStretch{};
std::array<float, 3> previousTranslation{};
std::array<float, 3> currentTranslation{};
Quaternion previousQuaternion{};
Quaternion currentQuaternion{};
float rotationAngle = 0.0f;
float inverseRotationSin = 0.0f;
bool linear = false;
bool identical = false;
bool valid = false;
@@ -426,6 +472,9 @@ struct PreparedTransformInterpolation {
// its matrices from a sibling, that is the sibling's partner.
size_t previousProjectionEntry = kNoPreparedPair;
bool indexedValid = false;
bool desktopIndexedValid = false;
// Unpaired quad held in camera space: its own transform is both endpoints.
bool cameraHold = false;
PreparedTransformInterpolation() {
indexedPairOffsets.fill(kNoPreparedPair);
@@ -433,7 +482,7 @@ struct PreparedTransformInterpolation {
};
PreparedAffinePair prepare_affine_pair(const Mat3x4<float>& previous,
const Mat3x4<float>& current) noexcept {
const Mat3x4<float>& current, bool rotatingDraw = false) noexcept {
PreparedAffinePair pair{.previous = previous, .current = current};
if (std::memcmp(&previous, &current, sizeof(current)) == 0) {
pair.identical = true;
@@ -443,10 +492,17 @@ PreparedAffinePair prepare_affine_pair(const Mat3x4<float>& previous,
std::array<std::array<float, 3>, 3> previousRotation{};
std::array<std::array<float, 3>, 3> currentRotation{};
if (!decompose_affine(previous, previousRotation, pair.previousScale,
pair.previousTranslation, pair.previousQuaternion) ||
!decompose_affine(current, currentRotation, pair.currentScale,
pair.currentTranslation, pair.currentQuaternion)) {
float previousSkew = 0.0f, currentSkew = 0.0f;
if (!decompose_affine(previous, previousRotation, pair.previousStretch,
pair.previousTranslation, pair.previousQuaternion, previousSkew) ||
!decompose_affine(current, currentRotation, pair.currentStretch,
pair.currentTranslation, pair.currentQuaternion, currentSkew)) {
return pair;
}
// Camera and seat anchors are rigid; a sheared one is garbage. Draws may shear:
// MKW's Lakitu sways by tilting his body's Y axis, and dropping that tilt from his
// rigid goggles (or holding them at the game frame) sank them into his skinned head.
if (!rotatingDraw && std::max(previousSkew, currentSkew) > 0.05f) {
return pair;
}
@@ -471,9 +527,15 @@ PreparedAffinePair prepare_affine_pair(const Mat3x4<float>& previous,
pair.currentQuaternion.w = -pair.currentQuaternion.w;
quaternionDot = -quaternionDot;
}
if (!std::isfinite(quaternionDot) || quaternionDot < kMinimumQuaternionDot) {
if (!std::isfinite(quaternionDot) || (!rotatingDraw && quaternionDot < kMinimumQuaternionDot)) {
return pair;
}
if (quaternionDot < kMinimumQuaternionDot) {
// A fast wheel is not a camera cut. Slerp its shortest arc so arbitrary VR
// sample weights keep angular speed constant without shrinking the wheel.
pair.rotationAngle = std::acos(std::clamp(quaternionDot, 0.0f, 1.0f));
pair.inverseRotationSin = 1.0f / std::sin(pair.rotationAngle);
}
pair.valid = true;
return pair;
}
@@ -539,10 +601,14 @@ bool evaluate_affine_pair(const PreparedAffinePair& pair, float weight,
return true;
}
// Consecutive 60 Hz transforms stay in one hemisphere and within 90 degrees, so
// normalized lerp is stable and skips three transcendentals per matrix.
const float previousWeight = 1.0f - weight;
const float currentWeight = weight;
// Most transforms use the cheaper normalized lerp. Fast rigid spins need
// constant angular speed across arbitrary display samples, so use slerp there.
const float previousWeight = pair.rotationAngle > 0.0f
? std::sin((1.0f - weight) * pair.rotationAngle) * pair.inverseRotationSin
: 1.0f - weight;
const float currentWeight = pair.rotationAngle > 0.0f
? std::sin(weight * pair.rotationAngle) * pair.inverseRotationSin
: weight;
Quaternion interpolated{
pair.previousQuaternion.x * previousWeight + pair.currentQuaternion.x * currentWeight,
pair.previousQuaternion.y * previousWeight + pair.currentQuaternion.y * currentWeight,
@@ -563,28 +629,23 @@ bool evaluate_affine_pair(const PreparedAffinePair& pair, float weight,
interpolated.w /= quaternionLength;
const auto interpolatedRotation = rotation_from_quaternion(interpolated);
std::array<float, 3> interpolatedScale{};
std::array<float, 6> u{};
for (size_t i = 0; i < u.size(); ++i) {
u[i] = pair.previousStretch[i] + (pair.currentStretch[i] - pair.previousStretch[i]) * weight;
}
std::array<float, 3> interpolatedTranslation{};
for (size_t i = 0; i < 3; ++i) {
interpolatedScale[i] = pair.previousScale[i] +
(pair.currentScale[i] - pair.previousScale[i]) * weight;
interpolatedTranslation[i] =
pair.previousTranslation[i] +
(pair.currentTranslation[i] - pair.previousTranslation[i]) * weight;
}
// Column scale mirrors decompose_affine: the reconstruction is R*S, with
// each scale component applied down its column.
output = {
{interpolatedRotation[0][0] * interpolatedScale[0],
interpolatedRotation[0][1] * interpolatedScale[1],
interpolatedRotation[0][2] * interpolatedScale[2], interpolatedTranslation[0]},
{interpolatedRotation[1][0] * interpolatedScale[0],
interpolatedRotation[1][1] * interpolatedScale[1],
interpolatedRotation[1][2] * interpolatedScale[2], interpolatedTranslation[1]},
{interpolatedRotation[2][0] * interpolatedScale[0],
interpolatedRotation[2][1] * interpolatedScale[1],
interpolatedRotation[2][2] * interpolatedScale[2], interpolatedTranslation[2]},
};
// Mirrors decompose_affine: the reconstruction is R*U, U upper-triangular.
const std::array<Vec4<float>*, 3> outputRows{&output.m0, &output.m1, &output.m2};
for (size_t row = 0; row < 3; ++row) {
const auto& r = interpolatedRotation[row];
*outputRows[row] = {r[0] * u[0], r[0] * u[1] + r[1] * u[3], r[0] * u[2] + r[1] * u[4] + r[2] * u[5],
interpolatedTranslation[row]};
}
return true;
}
@@ -619,6 +680,11 @@ bool interpolate_transform_midpoint(const Mat3x4<float>& previous, const Mat3x4<
return interpolate_transform(previous, current, 0.5f, output);
}
bool interpolate_draw_transform(const Mat3x4<float>& previous, const Mat3x4<float>& current,
float weight, Mat3x4<float>& output) noexcept {
return evaluate_affine_pair(prepare_affine_pair(previous, current, true), std::clamp(weight, 0.0f, 1.0f), output);
}
bool interpolate_indexed_transform(const Mat3x4<float>& previous,
const Mat3x4<float>& current, float weight,
Mat3x4<float>& output) noexcept {
@@ -677,6 +743,23 @@ Mat3x4<float> extrapolate_transform(const Mat3x4<float>& previous,
}
return predicted;
}
std::array<float, 3> translation_of(const Mat3x4<float>& matrix) noexcept {
return {matrix.m0.w(), matrix.m1.w(), matrix.m2.w()};
}
std::array<float, 3> rotate_vector(const Mat3x4<float>& matrix, const std::array<float, 3>& vector) noexcept {
const Vec4<float>* rows[] = {&matrix.m0, &matrix.m1, &matrix.m2};
std::array<float, 3> result{};
for (size_t row = 0; row < 3; ++row)
result[row] = (*rows[row])[0] * vector[0] + (*rows[row])[1] * vector[1] + (*rows[row])[2] * vector[2];
return result;
}
float distance_squared(const std::array<float, 3>& a, const std::array<float, 3>& b) noexcept {
const std::array<float, 3> delta{a[0] - b[0], a[1] - b[1], a[2] - b[2]};
return dot3(delta, delta);
}
} // namespace
float snapshot_match_distance_squared(const FrameTransformEntry& previousEntry,
@@ -789,7 +872,17 @@ void report_producer_paced(bool paced) noexcept {
s_pacingWindowMisses.store(0, std::memory_order_release);
}
void set_frame_interpolation_view_rebase(const Mat3x4<float>* currentFromPrevious,
const Mat3x4<float>* previousFromCurrent) noexcept {
s_rebaseView = currentFromPrevious && previousFromCurrent && stereo_frame_interpolation_active();
if (s_rebaseView) {
s_currentFromPreviousView = *currentFromPrevious;
s_previousFromCurrentView = *previousFromCurrent;
}
}
void begin_frame_interpolation() noexcept {
s_rebaseView = false;
const uint32_t targetFps = frame_interpolation_fps();
static bool previousStereo = false;
const bool stereo = stereo_frame_interpolation_active();
@@ -811,13 +904,17 @@ void begin_frame_interpolation() noexcept {
// pay thousands of small allocations every frame.
clear_index_map_keep_nodes(s_previousTransformIndices, s_previousFrameTransforms.size());
clear_index_map_keep_nodes(s_previousStableTransformIndices, s_previousFrameTransforms.size());
clear_index_map_keep_nodes(s_previousGeometryTransformIndices, s_previousFrameTransforms.size());
for (size_t i = 0; i < s_previousFrameTransforms.size(); ++i) {
const auto& identity = s_previousFrameTransforms[i].identity;
s_previousTransformIndices[identity.combined].push_back(i);
s_previousStableTransformIndices[stable_identity(identity)].push_back(i);
if (identity.geometry != 0)
s_previousGeometryTransformIndices[geometry_identity(identity)].push_back(i);
}
clear_index_map_keep_nodes(s_currentTransformIndices, s_previousFrameTransforms.size());
clear_index_map_keep_nodes(s_currentStableTransformIndices, s_previousFrameTransforms.size());
clear_index_map_keep_nodes(s_currentGeometryTransformIndices, s_previousFrameTransforms.size());
s_pendingUniformInterpolations.clear();
s_perspectiveCandidates = 0;
s_perspectiveMatchable = 0;
@@ -827,6 +924,34 @@ void begin_frame_interpolation() noexcept {
}
void finalize_frame_interpolation() noexcept {
s_diagPreparedDraws.store(0, std::memory_order_relaxed);
s_diagRejectedDraws.store(0, std::memory_order_relaxed);
s_diagVertexMotionDraws.store(0, std::memory_order_relaxed);
s_diagVertexMotionHeld.store(0, std::memory_order_relaxed);
if (s_rebaseView) {
// The old history retires at the end of this seal. Rebase it once, before
// instance matching and motion gates, rather than treating camera rotation
// around a distant object as an object teleport.
for (auto& entry : s_previousFrameTransforms) {
auto& transform = entry.transform;
if (transform.viewSpaceVertices) continue;
if (transform.indexedMatrices) {
for (size_t slot = 0; slot < MaxPnMtx; ++slot) {
if ((transform.usedMatrixMask & (1u << slot)) == 0) continue;
auto& matrices = *transform.indexedMatrices;
matrices.position[slot] = gfx::stereo_replay::compose_affine(s_currentFromPreviousView, matrices.position[slot]);
matrices.normal[slot] = gfx::stereo_replay::compose_normal(s_currentFromPreviousView, matrices.normal[slot]);
matrices.slotHash[slot] = xxh3_hash_s(&matrices.position[slot], sizeof(Mat3x4<float>),
xxh3_hash_s(&matrices.normal[slot], sizeof(Mat3x4<float>)));
}
} else {
transform.position = gfx::stereo_replay::compose_affine(s_currentFromPreviousView, transform.position);
transform.normal = gfx::stereo_replay::compose_normal(s_currentFromPreviousView, transform.normal);
if (entry.hasPrediction)
entry.predictedPosition = gfx::stereo_replay::compose_affine(s_currentFromPreviousView, entry.predictedPosition);
}
}
}
// A frame reported late seals without inserted slots, so the encode phase renders
// the native frame only. Its transforms still seed the next frame's matching.
const bool late = s_dropInterpolationAtSeal.exchange(false, std::memory_order_acq_rel);
@@ -893,6 +1018,210 @@ void finalize_frame_interpolation() noexcept {
static_cast<uint64_t>(cz);
};
// Vertex-motion quads come from particle emitters, which draw many look-alike
// quads: a speed line is two crossed quads, two new ones start on the same ring
// every frame, and each moves further per frame than the gap to its neighbours.
// Nearest-centre pairing swaps them and every swap sweeps a quad across the view,
// so these groups keep only pairs that are unambiguous in centre and shape.
struct QuadPrevious {
std::array<float, 3> world{}, camera{}, predicted{};
std::array<std::array<float, 3>, 2> worldEdges{}, cameraEdges{};
float size2 = 0.f, step2 = 0.f;
bool tracked = false;
};
struct QuadCurrent {
std::array<float, 3> center{};
std::array<std::array<float, 3>, 2> edges{};
float size2 = 0.f, shortEdge2 = 0.f;
};
struct QuadCandidate {
float cost = 0.f;
uint32_t previous = 0, current = 0;
};
struct QuadBest {
float cost = std::numeric_limits<float>::infinity();
float runnerUp = std::numeric_limits<float>::infinity();
uint32_t index = UINT32_MAX;
};
static std::vector<QuadPrevious> quadPrevious;
static std::vector<QuadCurrent> quadCurrent;
static std::vector<QuadCandidate> quadCandidates;
static std::vector<QuadBest> quadBestForCurrent, quadBestForPrevious;
static std::array<std::vector<uint32_t>, 2> quadPairs;
// A first pair that only one reading supports seeds the particle's path but is
// drawn held; it moves once the next frame lands on the prediction.
static std::vector<uint8_t> quadSeedOnly;
quadSeedOnly.assign(s_currentFrameTransforms.size(), 0);
const auto samePlace = [](const std::array<float, 3>& a, const std::array<float, 3>& b, float size2) noexcept {
return distance_squared(a, b) <= kQuadSamePlace * (size2 + 1.f);
};
const auto buildQuadCandidates = [&](bool followsCamera) {
quadCandidates.clear();
for (uint32_t current = 0; current < quadCurrent.size(); ++current) {
const auto& quad = quadCurrent[current];
for (uint32_t previous = 0; previous < quadPrevious.size(); ++previous) {
const auto& before = quadPrevious[previous];
// `<=` also drops a NaN delta.
if (!(distance_squared(before.world, quad.center) <=
kMaximumTranslationPerFrame * kMaximumTranslationPerFrame)) {
continue;
}
const auto shapeDistance = [&](const std::array<std::array<float, 3>, 2>& edges) noexcept {
return distance_squared(edges[0], quad.edges[0]) + distance_squared(edges[1], quad.edges[1]);
};
float centerCost, shapeCost;
if (before.tracked) {
centerCost = distance_squared(before.predicted, quad.center);
if (!(centerCost <= std::max(kQuadPathTolerance * before.step2, kQuadPathFloor * quad.shortEdge2))) {
continue;
}
shapeCost = std::min(shapeDistance(before.worldEdges), shapeDistance(before.cameraEdges));
} else {
centerCost = distance_squared(followsCamera ? before.camera : before.world, quad.center);
shapeCost = shapeDistance(followsCamera ? before.cameraEdges : before.worldEdges);
}
const float cost = centerCost + shapeCost;
if (!(shapeCost <= kQuadShapeChange * quad.size2) || !std::isfinite(cost)) {
continue;
}
quadCandidates.push_back({cost, previous, current});
}
}
};
// Keeps a pair only when each side prefers the other and beats its runner-up
// among other particles; the two quads of one cross are not each other's rivals.
const auto selectQuadPairs = [&](std::vector<uint32_t>& pairs) {
quadBestForCurrent.assign(quadCurrent.size(), {});
quadBestForPrevious.assign(quadPrevious.size(), {});
for (const auto& candidate : quadCandidates) {
auto& forCurrent = quadBestForCurrent[candidate.current];
if (candidate.cost < forCurrent.cost) {
forCurrent.cost = candidate.cost;
forCurrent.index = candidate.previous;
}
auto& forPrevious = quadBestForPrevious[candidate.previous];
if (candidate.cost < forPrevious.cost) {
forPrevious.cost = candidate.cost;
forPrevious.index = candidate.current;
}
}
for (const auto& candidate : quadCandidates) {
auto& forCurrent = quadBestForCurrent[candidate.current];
const auto& rival = quadPrevious[candidate.previous];
if (!samePlace(rival.world, quadPrevious[forCurrent.index].world, rival.size2)) {
forCurrent.runnerUp = std::min(forCurrent.runnerUp, candidate.cost);
}
auto& forPrevious = quadBestForPrevious[candidate.previous];
const auto& other = quadCurrent[candidate.current];
if (!samePlace(other.center, quadCurrent[forPrevious.index].center, other.size2)) {
forPrevious.runnerUp = std::min(forPrevious.runnerUp, candidate.cost);
}
}
pairs.assign(quadCurrent.size(), UINT32_MAX);
for (uint32_t current = 0; current < quadCurrent.size(); ++current) {
const auto& forCurrent = quadBestForCurrent[current];
if (forCurrent.index == UINT32_MAX) {
continue;
}
const auto& forPrevious = quadBestForPrevious[forCurrent.index];
if (samePlace(quadCurrent[current].center, quadCurrent[forPrevious.index].center,
quadCurrent[current].size2) &&
forCurrent.cost * kQuadAmbiguity < forCurrent.runnerUp &&
forPrevious.cost * kQuadAmbiguity < forPrevious.runnerUp) {
pairs[current] = forCurrent.index;
}
}
};
const auto matchQuadGroup = [&] {
const size_t previousCount = groupPreviousIndices.size();
const size_t currentCount = groupCurrentIndices.size();
if (previousCount * currentCount > kMaximumQuadPairs) {
return;
}
quadPrevious.resize(previousCount);
for (size_t index = 0; index < previousCount; ++index) {
const auto& entry = s_previousFrameTransforms[groupPreviousIndices[index]];
auto& quad = quadPrevious[index];
quad.world = translation_of(entry.transform.position);
quad.camera = entry.transform.quadCenter;
quad.cameraEdges = entry.transform.quadEdges;
for (size_t edge = 0; edge < 2; ++edge) {
quad.worldEdges[edge] =
s_rebaseView ? rotate_vector(s_currentFromPreviousView, quad.cameraEdges[edge]) : quad.cameraEdges[edge];
}
quad.size2 = dot3(quad.cameraEdges[0], quad.cameraEdges[0]) + dot3(quad.cameraEdges[1], quad.cameraEdges[1]);
quad.tracked = entry.hasPrediction;
if (quad.tracked) {
quad.predicted = translation_of(entry.predictedPosition);
quad.step2 = distance_squared(quad.predicted, quad.world);
}
}
quadCurrent.resize(currentCount);
for (size_t index = 0; index < currentCount; ++index) {
const auto& transform = s_currentFrameTransforms[groupCurrentIndices[index]].transform;
auto& quad = quadCurrent[index];
quad.center = translation_of(transform.position);
quad.edges = transform.quadEdges;
const float edge0 = dot3(quad.edges[0], quad.edges[0]);
const float edge1 = dot3(quad.edges[1], quad.edges[1]);
quad.size2 = edge0 + edge1;
quad.shortEdge2 = std::min(edge0, edge1);
}
// A particle without a path is read two ways: fixed in the world (smoke left
// behind) or carried with the camera (speed lines follow the kart). Without a
// camera rebase the two readings coincide.
const size_t readings = s_rebaseView ? 2 : 1;
for (size_t reading = 0; reading < readings; ++reading) {
buildQuadCandidates(reading == 1);
selectQuadPairs(quadPairs[reading]);
}
// One emitter's particles move alike. Particles with a path show which reading
// fits: carried ones step less in camera space than in the world.
bool followsCamera = false;
if (readings == 2) {
double worldSteps = 0.0, cameraSteps = 0.0;
uint32_t tracked = 0, worldFirstSteps = 0, cameraFirstSteps = 0;
for (size_t current = 0; current < currentCount; ++current) {
if (const uint32_t previous = quadPairs[0][current]; previous != UINT32_MAX) {
const auto& before = quadPrevious[previous];
if (before.tracked) {
worldSteps += distance_squared(before.world, quadCurrent[current].center);
cameraSteps += distance_squared(before.camera, quadCurrent[current].center);
++tracked;
} else {
++worldFirstSteps;
}
}
if (const uint32_t previous = quadPairs[1][current];
previous != UINT32_MAX && !quadPrevious[previous].tracked) {
++cameraFirstSteps;
}
}
followsCamera = tracked != 0 ? cameraSteps < worldSteps : cameraFirstSteps > worldFirstSteps;
}
const auto& pairs = quadPairs[followsCamera ? 1 : 0];
const auto& otherPairs = quadPairs[followsCamera ? 0 : 1];
for (size_t current = 0; current < currentCount; ++current) {
const size_t currentIndex = groupCurrentIndices[current];
auto& transform = s_currentFrameTransforms[currentIndex].transform;
const uint32_t previous = pairs[current];
// A first step shows only when both readings choose it. A newborn that spawns
// where the last one did fools one reading, rarely both.
const bool seedOnly = previous != UINT32_MAX && !quadPrevious[previous].tracked &&
!(readings == 2 && otherPairs[current] == previous);
transform.holdInCamera = followsCamera && (previous == UINT32_MAX || seedOnly);
if (previous == UINT32_MAX) {
continue;
}
const size_t previousIndex = groupPreviousIndices[previous];
currentToPrevious[currentIndex] = previousIndex;
currentMatched[currentIndex] = 1;
previousMatched[previousIndex] = 1;
quadSeedOnly[currentIndex] = seedOnly;
}
};
const auto matchGroups =
[&](const auto& currentGroups, const auto& previousGroups,
bool allowOrderedFallback) {
@@ -919,6 +1248,12 @@ void finalize_frame_interpolation() noexcept {
if (groupCurrentIndices.empty() || groupPreviousIndices.empty()) {
continue;
}
// Draw identity includes the vertex-motion flag, so a group is all quads or none.
// One quad on each side is no proof either: a particle died and another spawned.
if (s_currentFrameTransforms[groupCurrentIndices.front()].transform.vertexMotion.enabled) {
matchQuadGroup();
continue;
}
// A unique draw has no identity ambiguity. Keep the conservative
// interpolation fallback for malformed/non-finite matrices.
@@ -1078,6 +1413,9 @@ void finalize_frame_interpolation() noexcept {
}
};
matchGroups(s_currentTransformIndices, s_previousTransformIndices, false);
// Prefer unchanged meshes across a texture flip before the material-only
// fallback for deforming geometry. Both passes keep palette topology strict.
matchGroups(s_currentGeometryTransformIndices, s_previousGeometryTransformIndices, false);
matchGroups(s_currentStableTransformIndices, s_previousStableTransformIndices, true);
s_perspectiveMatches = static_cast<uint32_t>(std::count_if(
@@ -1097,7 +1435,7 @@ void finalize_frame_interpolation() noexcept {
s_diagMatches.store(s_perspectiveMatches, std::memory_order_relaxed);
s_diagEligible.store(eligible, std::memory_order_relaxed);
s_diagReplaySafe.store(replaySafe, std::memory_order_relaxed);
if (eligible) {
if (eligible || stereo_frame_interpolation_active()) {
s_diagFramesSealed.fetch_add(1, std::memory_order_relaxed);
if (s_perspectiveMatchable != 0 &&
s_perspectiveMatches * 100 < s_perspectiveMatchable * kLowMatchPercent) {
@@ -1113,9 +1451,11 @@ void finalize_frame_interpolation() noexcept {
// identity encoded by PNMTXIDX and are prepared as one coherent draw below.
constexpr float kMaximumPredictionSeedDeltaSquared =
kMaximumTranslationPerFrame * kMaximumTranslationPerFrame;
uint32_t quadsHeld = 0;
for (size_t currentIndex = 0; currentIndex < currentToPrevious.size(); ++currentIndex) {
const size_t previousIndex = currentToPrevious[currentIndex];
if (previousIndex == SIZE_MAX) {
quadsHeld += s_currentFrameTransforms[currentIndex].transform.vertexMotion.enabled;
continue;
}
auto& currentEntry = s_currentFrameTransforms[currentIndex];
@@ -1123,6 +1463,41 @@ void finalize_frame_interpolation() noexcept {
if (currentEntry.transform.indexedMatrices || previousEntry.transform.indexedMatrices) {
continue;
}
// A looping rigid animation can wrap by far less than the teleport limit
// (Coconut Mall resets its escalator phase every 20 local units). Blending
// that jump produces a brief reverse sweep. Only cut a large, nearly
// opposite jump after a measured velocity, with the same mesh and basis.
// Camera motion has already been removed above. Particle births are not
// cyclic rigid animations and must not use this test.
if (s_rebaseView && previousEntry.hasPrediction &&
!currentEntry.transform.vertexMotion.enabled && currentEntry.identity.geometry != 0 &&
currentEntry.identity.geometry == previousEntry.identity.geometry) {
const auto& before = previousEntry.transform.position;
const auto& now = currentEntry.transform.position;
const auto& predicted = previousEntry.predictedPosition;
const Vec4<float>* oldRows[] = {&before.m0, &before.m1, &before.m2};
const Vec4<float>* newRows[] = {&now.m0, &now.m1, &now.m2};
const Vec4<float>* predictedRows[] = {&predicted.m0, &predicted.m1, &predicted.m2};
float speed2 = 0.f, jump2 = 0.f, dot = 0.f, basisDelta = 0.f, basisSize = 0.f;
for (size_t row = 0; row < 3; ++row) {
const float velocity = (*predictedRows[row])[3] - (*oldRows[row])[3];
const float delta = (*newRows[row])[3] - (*oldRows[row])[3];
speed2 += velocity * velocity;
jump2 += delta * delta;
dot += velocity * delta;
for (size_t col = 0; col < 3; ++col) {
const float difference = (*newRows[row])[col] - (*oldRows[row])[col];
basisDelta += difference * difference;
basisSize += (*newRows[row])[col] * (*newRows[row])[col];
}
}
if (speed2 > 0.0001f && jump2 > std::max(1.f, 16.f * speed2) && dot < 0.f &&
dot * dot > 0.9f * speed2 * jump2 && basisDelta < 0.0001f * basisSize) {
currentToPrevious[currentIndex] = SIZE_MAX;
s_diagAnimationWrapCuts.fetch_add(1, std::memory_order_relaxed);
continue; // Do not seed the next frame with the reset's apparent velocity.
}
}
// Seed the next frame's matching with a constant-velocity reference, but never from
// a pair the interpolator would reject as a teleport (`<=` so NaN fails too).
if (translation_delta_squared(previousEntry.transform.position,
@@ -1132,7 +1507,13 @@ void finalize_frame_interpolation() noexcept {
currentEntry.transform.position);
currentEntry.hasPrediction = true;
}
if (quadSeedOnly[currentIndex]) {
// Seeded above, drawn held: it moves once the next frame lands on the path.
currentToPrevious[currentIndex] = SIZE_MAX;
++quadsHeld;
}
}
s_diagVertexMotionHeld.store(quadsHeld, std::memory_order_relaxed);
if (eligible || stereo_frame_interpolation_active()) {
// Prepare each matched pair once: every sample of a draw shares the same
@@ -1140,18 +1521,38 @@ void finalize_frame_interpolation() noexcept {
std::vector<PreparedTransformInterpolation> preparedTransforms(s_currentFrameTransforms.size());
std::vector<uint8_t> preparedTransformState(s_currentFrameTransforms.size(), 0);
std::vector<PreparedAffinePair> preparedPairs;
std::vector<PreparedAffinePair> desktopPairs;
const bool separateDesktopPairs = s_rebaseView && eligible;
preparedPairs.reserve(s_pendingUniformInterpolations.size() * 2);
const auto appendPreparedPair = [&](const Mat3x4<float>& previousPosition,
const Mat3x4<float>& currentPosition,
const Mat3x4<float>& previousNormal,
const Mat3x4<float>& currentNormal,
bool indexed) {
bool indexed, bool rebased = true, bool vertexMotion = false) {
const size_t pairOffset = preparedPairs.size();
const auto retainCurrentBasis = [&](const Mat3x4<float>& position) {
if (!vertexMotion) return position;
auto result = currentPosition;
result.m0[3] = position.m0[3];
result.m1[3] = position.m1[3];
result.m2[3] = position.m2[3];
return result;
};
if (separateDesktopPairs) {
const auto originalPosition = rebased ? gfx::stereo_replay::compose_affine(s_previousFromCurrentView, previousPosition)
: previousPosition;
const auto originalNormal = rebased ? gfx::stereo_replay::compose_normal(s_previousFromCurrentView, previousNormal)
: previousNormal;
desktopPairs.push_back(indexed ? prepare_indexed_pair(originalPosition, currentPosition)
: prepare_affine_pair(retainCurrentBasis(originalPosition), currentPosition, true));
desktopPairs.push_back(indexed ? prepare_indexed_pair(originalNormal, currentNormal)
: prepare_affine_pair(vertexMotion ? currentNormal : originalNormal, currentNormal, true));
}
preparedPairs.push_back(indexed ? prepare_indexed_pair(previousPosition, currentPosition)
: prepare_affine_pair(previousPosition, currentPosition));
: prepare_affine_pair(retainCurrentBasis(previousPosition), currentPosition, true));
preparedPairs.push_back(indexed ? prepare_indexed_pair(previousNormal, currentNormal)
: prepare_affine_pair(previousNormal, currentNormal));
: prepare_affine_pair(vertexMotion ? currentNormal : previousNormal, currentNormal, true));
return pairOffset;
};
@@ -1276,6 +1677,7 @@ void finalize_frame_interpolation() noexcept {
// A palette is one deformation unit: interpolating only the resolved slots cracks
// the mesh, so any unresolved slot duplicates the whole current draw.
bool allSlotsValid = true;
bool desktopSlotsValid = true;
for (size_t slot = 0; slot < MaxPnMtx; ++slot) {
if ((current.usedMatrixMask & (1u << slot)) == 0) {
continue;
@@ -1283,6 +1685,7 @@ void finalize_frame_interpolation() noexcept {
const auto& resolved = resolvedSlots[static_cast<size_t>(palette) * MaxPnMtx + slot];
if (resolved.position == nullptr) {
allSlotsValid = false;
desktopSlotsValid = false;
break;
}
const size_t pairOffset =
@@ -1291,22 +1694,58 @@ void finalize_frame_interpolation() noexcept {
prepared.indexedPairOffsets[slot] = pairOffset;
if (!preparedPairs[pairOffset].valid || !preparedPairs[pairOffset + 1].valid) {
allSlotsValid = false;
break;
}
const auto& desktop = separateDesktopPairs ? desktopPairs : preparedPairs;
desktopSlotsValid &= desktop[pairOffset].valid && desktop[pairOffset + 1].valid;
}
prepared.indexedValid = allSlotsValid;
prepared.desktopIndexedValid = desktopSlotsValid;
} else {
const size_t previousTransformIndex = currentToPrevious[task.currentTransformIndex];
if (previousTransformIndex >= s_previousFrameTransforms.size()) {
if (current.holdInCamera && s_rebaseView) {
// As if it sat at the same camera-space place last frame: the sampled
// camera then carries it, like the emitter it follows.
prepared.cameraHold = true;
prepared.nonIndexedPairOffset = appendPreparedPair(
gfx::stereo_replay::compose_affine(s_currentFromPreviousView, current.position), current.position,
current.normal, current.normal, false, true, true);
}
continue;
}
const auto& previous = s_previousFrameTransforms[previousTransformIndex].transform;
prepared.previousProjectionEntry = previousTransformIndex;
prepared.nonIndexedPairOffset = appendPreparedPair(
previous.position, current.position, previous.normal, current.normal, false);
previous.position, current.position, previous.normal, current.normal, false, !previous.viewSpaceVertices,
current.vertexMotion.enabled);
}
}
uint32_t preparedDraws = 0;
uint32_t rejectedDraws = 0;
uint32_t vertexMotionDraws = 0;
for (size_t i = 0; i < preparedTransforms.size(); ++i) {
if (preparedTransformState[i] == 0)
continue;
const auto& prepared = preparedTransforms[i];
if (prepared.previousProjectionEntry == kNoPreparedPair)
continue;
const bool valid = s_currentFrameTransforms[i].transform.indexedMatrices
? prepared.indexedValid
: prepared.nonIndexedPairOffset != kNoPreparedPair &&
preparedPairs[prepared.nonIndexedPairOffset].valid &&
preparedPairs[prepared.nonIndexedPairOffset + 1].valid;
if (valid)
++preparedDraws;
else
++rejectedDraws;
if (valid && s_currentFrameTransforms[i].transform.vertexMotion.enabled)
++vertexMotionDraws;
}
s_diagPreparedDraws.store(preparedDraws, std::memory_order_relaxed);
s_diagRejectedDraws.store(rejectedDraws, std::memory_order_relaxed);
s_diagVertexMotionDraws.store(vertexMotionDraws, std::memory_order_relaxed);
const auto interpolatePendingUniform = [&](const auto& task) {
if (task.currentTransformIndex >= s_currentFrameTransforms.size()) {
return;
@@ -1315,20 +1754,24 @@ void finalize_frame_interpolation() noexcept {
const auto& current = s_currentFrameTransforms[task.currentTransformIndex].transform;
std::memcpy(task.uniformData, task.sourceUniformData, task.uniformSize);
const auto& prepared = preparedTransforms[task.currentTransformIndex];
if (task.indexedMatrices && !prepared.indexedValid) {
const bool desktopSample = task.numerator != 0;
const auto& pairs = desktopSample && separateDesktopPairs ? desktopPairs : preparedPairs;
if (task.indexedMatrices && !(desktopSample ? prepared.desktopIndexedValid : prepared.indexedValid)) {
return;
}
// The projection comes from whichever previous entry supplied the transforms, which
// for a borrowed palette is a sibling's partner.
// for a borrowed palette is a sibling's partner. A camera hold has no partner.
const size_t previousTransformIndex = prepared.previousProjectionEntry;
if (previousTransformIndex >= s_previousFrameTransforms.size()) {
if (previousTransformIndex >= s_previousFrameTransforms.size() && !prepared.cameraHold) {
return;
}
const auto& previous = s_previousFrameTransforms[previousTransformIndex].transform;
const auto& previousProjection = prepared.cameraHold
? current.projection
: s_previousFrameTransforms[previousTransformIndex].transform.projection;
const float weight =
static_cast<float>(task.numerator) / static_cast<float>(task.denominator);
const auto interpolatedProjection =
interpolate_projection(previous.projection, current.projection, weight);
interpolate_projection(previousProjection, current.projection, weight);
std::memcpy(task.uniformData + task.projectionOffset, &interpolatedProjection,
sizeof(interpolatedProjection));
@@ -1338,8 +1781,10 @@ void finalize_frame_interpolation() noexcept {
}
Mat3x4<float> interpolatedPosition{};
Mat3x4<float> interpolatedNormal{};
evaluate_affine_pair(preparedPairs[pairOffset], weight, interpolatedPosition);
evaluate_affine_pair(preparedPairs[pairOffset + 1], weight, interpolatedNormal);
evaluate_affine_pair(pairs[pairOffset], weight, interpolatedPosition);
evaluate_affine_pair(pairs[pairOffset + 1], weight, interpolatedNormal);
if (current.vertexMotion.enabled && !task.indexedMatrices)
interpolatedPosition = offset_transform_origin(interpolatedPosition, current.vertexMotion.center, -1.f);
std::memcpy(task.uniformData + task.positionOffset + currentIndex * sizeof(Mat3x4<float>),
&interpolatedPosition, sizeof(interpolatedPosition));
std::memcpy(task.uniformData + task.normalOffset + currentIndex * sizeof(Mat3x4<float>),
@@ -1393,6 +1838,11 @@ void get_frame_interpolation_diagnostics(AuroraFrameInterpolationDiagnostics& di
diagnostics.framesReplayUnsafe = s_diagFramesReplayUnsafe.load(std::memory_order_relaxed);
diagnostics.slotReductions = s_diagSlotReductions.load(std::memory_order_relaxed);
diagnostics.lateSealDrops = s_diagLateSealDrops.load(std::memory_order_relaxed);
diagnostics.preparedDraws = s_diagPreparedDraws.load(std::memory_order_relaxed);
diagnostics.rejectedDraws = s_diagRejectedDraws.load(std::memory_order_relaxed);
diagnostics.vertexMotionDraws = s_diagVertexMotionDraws.load(std::memory_order_relaxed);
diagnostics.vertexMotionHeld = s_diagVertexMotionHeld.load(std::memory_order_relaxed);
diagnostics.animationWrapCuts = s_diagAnimationWrapCuts.load(std::memory_order_relaxed);
}
bool has_interpolated_frame() noexcept {
@@ -1463,6 +1913,7 @@ std::array<gfx::Range, MaxInterpolatedFrames> record_interpolation_draw(const Fr
FrameTransformSnapshot snapshot{
.projection = projection,
.usedMatrixMask = usedPnMtxMask,
.vertexMotion = uniformLayout.vertexMotion,
};
if (uniformLayout.indexedMatrices) {
snapshot.indexedMatrices = acquire_indexed_matrices();
@@ -1482,6 +1933,23 @@ std::array<gfx::Range, MaxInterpolatedFrames> record_interpolation_draw(const Fr
const size_t currentMatrix = std::min<size_t>(g_gxState.currentPnMtx, MaxPnMtx - 1);
snapshot.position = g_gxState.pnMtx[currentMatrix].pos;
snapshot.normal = g_gxState.pnMtx[currentMatrix].nrm;
if (uniformLayout.vertexMotion.enabled) {
snapshot.position = offset_transform_origin(snapshot.position, uniformLayout.vertexMotion.center);
const Vec4<float>* rows[] = {&snapshot.position.m0, &snapshot.position.m1, &snapshot.position.m2};
const auto& shape = uniformLayout.vertexShape;
for (size_t row = 0; row < 3; ++row) {
const auto& basis = *rows[row];
snapshot.quadCenter[row] = basis[3];
snapshot.quadEdges[0][row] = basis[0] * shape.edge0[0] + basis[1] * shape.edge0[1] + basis[2] * shape.edge0[2];
snapshot.quadEdges[1][row] = basis[0] * shape.edge1[0] + basis[1] * shape.edge1[1] + basis[2] * shape.edge1[2];
}
} else if (g_gxState.vtxDesc[GX_VA_POS] == GX_DIRECT) {
const Vec4<float>* rows[] = {&snapshot.position.m0, &snapshot.position.m1, &snapshot.position.m2};
snapshot.viewSpaceVertices = true;
for (size_t row = 0; row < 3; ++row)
for (size_t col = 0; col < 4; ++col)
snapshot.viewSpaceVertices &= (*rows[row])[col] == (row == col ? 1.f : 0.f);
}
}
const size_t currentTransformIndex = s_currentFrameTransforms.size();
@@ -1492,14 +1960,20 @@ std::array<gfx::Range, MaxInterpolatedFrames> record_interpolation_draw(const Fr
s_currentTransformIndices[identity.combined].push_back(currentTransformIndex);
const HashType stableIdentity = stable_identity(identity);
s_currentStableTransformIndices[stableIdentity].push_back(currentTransformIndex);
if (identity.geometry != 0)
s_currentGeometryTransformIndices[geometry_identity(identity)].push_back(currentTransformIndex);
std::array<gfx::Range, MaxInterpolatedFrames> interpolatedRanges{};
++s_perspectiveCandidates;
const auto exactPrevious = s_previousTransformIndices.find(identity.combined);
const auto stablePrevious = s_previousStableTransformIndices.find(stableIdentity);
const auto geometryPrevious = identity.geometry != 0
? s_previousGeometryTransformIndices.find(geometry_identity(identity))
: s_previousGeometryTransformIndices.end();
const bool hasPreviousPartner =
(exactPrevious != s_previousTransformIndices.end() && !exactPrevious->second.empty()) ||
(stablePrevious != s_previousStableTransformIndices.end() && !stablePrevious->second.empty());
(stablePrevious != s_previousStableTransformIndices.end() && !stablePrevious->second.empty()) ||
(geometryPrevious != s_previousGeometryTransformIndices.end() && !geometryPrevious->second.empty());
if (hasPreviousPartner) {
++s_perspectiveMatchable;
}
+40 -2
View File
@@ -11,8 +11,8 @@
namespace aurora::gx {
constexpr uint32_t MaxInterpolatedFrames = 3;
// Identifies a draw across frames. `combined` includes the geometry signature and is exact, while
// `pipeline`/`texture` is the material-only fallback for meshes whose vertex data changes.
// Identifies a draw across frames. Match exact `combined` first, then `pipeline`/`geometry`
// across texture animation, then `pipeline`/`texture` for meshes whose vertex data changes.
struct FrameInterpolationDrawIdentity {
HashType combined = 0;
HashType pipeline = 0;
@@ -20,8 +20,35 @@ struct FrameInterpolationDrawIdentity {
// Hash of the per-vertex PNMTXIDX stream. Matrix values animate, but this topology decides which
// absolute palette slot each vertex reads, so it must match across a pair.
HashType matrixTopology = 0;
// Exact geometry without its texture binding. Texture-pattern animations may
// swap images while the same mesh still needs continuous camera/object motion.
// Zero means unavailable; do not use it as a wildcard geometry match.
HashType geometry = 0;
};
// CPU-authored particle quads move inside their vertex stream, often with an
// identity position matrix. Track their center while retaining the current
// shape, UVs and colour; no renderer thread needs access to guest particles.
struct DrawVertexMotion {
std::array<float, 3> center{};
bool enabled = false;
};
// Matching-only shape of such a quad: the two edges leaving its first corner.
// Emitters draw many look-alike quads, and a speed line moves further per
// frame than the gap to its neighbours; its size and orientation still tell
// them apart. Replay never needs it, so it stays out of UniformReplayLayout.
struct DrawVertexShape {
std::array<float, 3> edge0{}, edge1{};
};
inline Mat3x4<float> offset_transform_origin(Mat3x4<float> matrix,
const std::array<float, 3>& center, float sign = 1.f) noexcept {
for (auto* row : {&matrix.m0, &matrix.m1, &matrix.m2})
(*row)[3] += sign * ((*row)[0] * center[0] + (*row)[1] * center[1] + (*row)[2] * center[2]);
return matrix;
}
// Where the transforms live inside a draw's staged uniform block. Offsets are
// relative to the start of the mapped range.
struct InterpolatedUniformLayout {
@@ -33,6 +60,8 @@ struct InterpolatedUniformLayout {
// Slot the live matrix occupies; a compacted position region holds it at 0.
size_t currentMatrix = 0;
bool indexedMatrices = false;
DrawVertexMotion vertexMotion{};
DrawVertexShape vertexShape{};
};
namespace detail {
@@ -61,6 +90,10 @@ void report_producer_paced(bool paced) noexcept;
void begin_frame_interpolation() noexcept;
void finalize_frame_interpolation() noexcept;
// Before seal, re-express previous VR endpoints in the current recorded camera.
// Both null disables the operation. begin_frame_interpolation clears it.
void set_frame_interpolation_view_rebase(const Mat3x4<float>* currentFromPrevious,
const Mat3x4<float>* previousFromCurrent) noexcept;
// Fills the observability snapshot behind aurora_get_frame_interpolation_diagnostics.
void get_frame_interpolation_diagnostics(AuroraFrameInterpolationDiagnostics& diagnostics) noexcept;
bool has_interpolated_frame() noexcept;
@@ -87,6 +120,11 @@ bool interpolate_transform(const Mat3x4<float>& previous, const Mat3x4<float>& c
float weight, Mat3x4<float>& output) noexcept;
bool interpolate_transform_midpoint(const Mat3x4<float>& previous, const Mat3x4<float>& current,
Mat3x4<float>& output) noexcept;
// Matched rigid draws can spin more than 90 degrees per guest frame (kart tires), and
// a draw matrix may shear (Lakitu's sway tilts his whole body). Camera/seat anchors
// keep the conservative rotation and rigidity guards above.
bool interpolate_draw_transform(const Mat3x4<float>& previous, const Mat3x4<float>& current,
float weight, Mat3x4<float>& output) noexcept;
// Matrix palettes are already-composed skinning transforms, so interpolating their coefficients
// keeps shared boundaries intact. Ordinary one-matrix draws keep the rigid TRS path above.
bool interpolate_indexed_transform(const Mat3x4<float>& previous,
+5 -1
View File
@@ -582,7 +582,8 @@ static Mat4x4<float> effective_projection() noexcept {
UniformRanges build_uniform(const ShaderInfo& info, u32 vtxStart, const BindGroupRanges& ranges,
const FrameInterpolationDrawIdentity& drawIdentity, bool perspective,
uint16_t usedPnMtxMask) noexcept {
uint16_t usedPnMtxMask, DrawVertexMotion vertexMotion,
const DrawVertexShape& vertexShape) noexcept {
ZoneScoped;
auto [buf, range] = gfx::map_uniform(info.uniformSize);
@@ -793,6 +794,7 @@ UniformRanges build_uniform(const ShaderInfo& info, u32 vtxStart, const BindGrou
.normalMatrixCount = layout.nrmCount,
.perspective = perspective,
.indexedMatrices = info.indexAttr.test(GX_VA_PNMTXIDX),
.vertexMotion = vertexMotion,
.nativeEfbEffect = nativeEfbEffect,
.compositeDepthCopy = nativeEfbEffect ? compositeDepthCopy : nullptr,
};
@@ -818,6 +820,8 @@ UniformRanges build_uniform(const ShaderInfo& info, u32 vtxStart, const BindGrou
// A compacted position region holds the current matrix at slot 0.
.currentMatrix = layout.absolutePosRegion ? std::min<size_t>(g_gxState.currentPnMtx, MaxPnMtx - 1) : 0,
.indexedMatrices = info.indexAttr.test(GX_VA_PNMTXIDX),
.vertexMotion = vertexMotion,
.vertexShape = vertexShape,
},
&previousUniform);
g_gxState.stateDirty = false;
+3 -1
View File
@@ -14,6 +14,7 @@ struct UniformReplayLayout {
uint8_t normalMatrixCount = 0;
bool perspective = false;
bool indexedMatrices = false;
DrawVertexMotion vertexMotion{};
// A 2D draw compositing the framebuffer back over itself: bloom, blur and the
// rest of the native post-processing chain. It belongs to the rendered image,
// not to the game's 2D layer, so it must stay where the game aimed it.
@@ -38,6 +39,7 @@ ShaderInfo build_shader_info(const ShaderConfig& config) noexcept;
Light prepare_shader_light(Light light) noexcept;
UniformRanges build_uniform(const ShaderInfo& info, uint32_t vtxStart, const BindGroupRanges& ranges,
const FrameInterpolationDrawIdentity& drawIdentity, bool perspective,
uint16_t usedPnMtxMask = 1) noexcept;
uint16_t usedPnMtxMask = 1, DrawVertexMotion vertexMotion = {},
const DrawVertexShape& vertexShape = {}) noexcept;
u8 color_channel(GXChannelID id) noexcept;
}; // namespace aurora::gx
+74
View File
@@ -0,0 +1,74 @@
#pragma once
#include "gx/frame_interpolation.hpp"
#include "gfx/stereo_replay.hpp"
#include <cstdint>
#include <cstring>
namespace aurora::stereo {
inline bool inverse_rigid_view(const Mat3x4<float>& view, Mat3x4<float>& inverse) noexcept {
const Vec4<float>* rows[] = {&view.m0, &view.m1, &view.m2};
for (unsigned i = 0; i < 3; ++i) {
for (unsigned j = 0; j < 4; ++j) {
// Hosts may build Aurora with finite-math-only.
uint32_t bits;
std::memcpy(&bits, reinterpret_cast<const uint8_t*>(rows[i]) + j * sizeof(float), sizeof(bits));
if ((bits & 0x7f800000u) == 0x7f800000u) return false;
}
for (unsigned j = 0; j < 3; ++j) {
float dot = 0;
for (unsigned k = 0; k < 3; ++k) dot += (*rows[i])[k] * (*rows[j])[k];
if (std::abs(dot - (i == j ? 1.f : 0.f)) > 0.002f) return false;
}
}
const float determinant = view.m0[0] * (view.m1[1] * view.m2[2] - view.m1[2] * view.m2[1]) -
view.m0[1] * (view.m1[0] * view.m2[2] - view.m1[2] * view.m2[0]) +
view.m0[2] * (view.m1[0] * view.m2[1] - view.m1[1] * view.m2[0]);
if (determinant < 0.99f) return false;
Vec4<float>* out[] = {&inverse.m0, &inverse.m1, &inverse.m2};
for (unsigned i = 0; i < 3; ++i) {
(*out[i])[3] = 0;
for (unsigned j = 0; j < 3; ++j) {
(*out[i])[j] = (*rows[j])[i];
(*out[i])[3] -= (*rows[j])[i] * (*rows[j])[3];
}
}
return true;
}
// Keep both object endpoints in the current recorded camera's coordinates.
// Sample the actual camera pose separately, so held vertices, unmatched draws
// and rejected object animations still follow smooth game-camera movement.
struct SceneCameraMotion {
Mat3x4<float> currentFromPrevious{};
Mat3x4<float> previousFromCurrent{};
Mat3x4<float> worldFromCurrentScene{};
Mat3x4<float> previousPose{}, currentPose{};
bool active = false;
bool prepare(const Mat3x4<float>& previousView, const Mat3x4<float>& currentView,
const Mat3x4<float>& previousAnchor, const Mat3x4<float>& currentAnchor) noexcept {
active = false;
Mat3x4<float> previousCameraPose{}, sample{};
if (!inverse_rigid_view(previousView, previousCameraPose) ||
!inverse_rigid_view(currentView, worldFromCurrentScene) ||
!gx::interpolate_transform(previousCameraPose, worldFromCurrentScene, 0.5f, sample) ||
!inverse_rigid_view(gfx::stereo_replay::compose_affine(previousAnchor, previousView), previousPose) ||
!inverse_rigid_view(gfx::stereo_replay::compose_affine(currentAnchor, currentView), currentPose) ||
!gx::interpolate_transform(previousPose, currentPose, 0.5f, sample)) return false;
currentFromPrevious = gfx::stereo_replay::compose_affine(currentView, previousCameraPose);
previousFromCurrent = gfx::stereo_replay::compose_affine(previousView, worldFromCurrentScene);
active = true;
return true;
}
bool sample(float weight, Mat3x4<float>& anchorFromCurrentScene) const noexcept {
Mat3x4<float> pose{}, view{};
if (!active || !gx::interpolate_transform(previousPose, currentPose, weight, pose) ||
!inverse_rigid_view(pose, view)) return false;
anchorFromCurrentScene = gfx::stereo_replay::compose_affine(view, worldFromCurrentScene);
return true;
}
};
} // namespace aurora::stereo
+89 -7
View File
@@ -4,14 +4,96 @@
#include <cstdint>
namespace aurora::stereo {
// The current scene belongs to the VI presentation boundary. Delaying scene
// motion by one guest interval lets every display sample fall between known
// endpoints; head tracking is applied afterwards at its predicted display time.
inline float interpolation_weight(uint64_t displayTime, uint64_t boundary, uint64_t interval) noexcept {
if (displayTime == 0 || boundary == 0 || interval == 0)
// Scene playback uses the producer's cadence on a local monotonic clock. The
// compositor may predict head poses several frames ahead; sampling the two known
// scene endpoints at that future time would clamp every image to the newest one.
// Anchor when history starts. If production later arrives earlier on that grid
// (e.g. after shader warm-up), advance the origin enough to reach the new pair.
// Never move playback backwards to follow a late seal or the prediction horizon.
class ScenePlaybackClock {
public:
void begin_scene(uint64_t boundary, uint64_t now, bool continuous) noexcept {
if (!continuous || originBoundary_ == 0 || now < originTime_ || boundary <= lastBoundary_ ||
sample_time(now) < boundary) {
originBoundary_ = boundary;
originTime_ = now;
}
lastBoundary_ = boundary;
}
uint64_t sample_time(uint64_t now) const noexcept {
if (originBoundary_ == 0 || now < originTime_)
return 0;
return originBoundary_ + (now - originTime_);
}
private:
uint64_t originBoundary_ = 0;
uint64_t originTime_ = 0;
uint64_t lastBoundary_ = 0;
};
// The current scene belongs to boundary, and playback blends from the preceding
// scene for one guest interval. Head tracking is applied afterwards, undelayed.
inline float interpolation_weight(uint64_t sampleTime, uint64_t boundary, uint64_t interval) noexcept {
if (sampleTime == 0 || boundary == 0 || interval == 0)
return 1.0f;
if (displayTime <= boundary)
if (sampleTime <= boundary)
return 0.0f;
return static_cast<float>(std::min(static_cast<double>(displayTime - boundary) / static_cast<double>(interval), 1.0));
return static_cast<float>(std::min(static_cast<double>(sampleTime - boundary) / static_cast<double>(interval), 1.0));
}
// Worker-owned window. A fresh eye pair can still repeat the same game-scene
// time, or move it backwards when an overdue scene is replaced. Neither is
// visible in the compositor's submission FPS. Discontinuities have no comparable
// scene time, so they break the sequence rather than counting as backwards motion.
struct MotionSamples {
uint32_t samples = 0;
uint32_t blended = 0;
uint32_t atPrevious = 0;
uint32_t atCurrent = 0;
uint32_t discontinuous = 0;
uint32_t repeated = 0;
uint32_t backwards = 0;
uint64_t lastSceneTime = 0;
uint64_t minStep = UINT64_MAX;
uint64_t maxStep = 0;
void record(uint64_t displayTime, uint64_t boundary, uint64_t interval,
bool continuous, float weight) noexcept {
++samples;
if (!continuous || displayTime == 0 || boundary < interval || interval == 0) {
++discontinuous;
lastSceneTime = 0;
return;
}
if (weight <= 0.0f)
++atPrevious;
else if (weight >= 1.0f)
++atCurrent;
else
++blended;
const uint64_t sceneTime = boundary - interval +
static_cast<uint64_t>(static_cast<double>(interval) * weight);
if (lastSceneTime != 0) {
if (sceneTime >= lastSceneTime) {
const auto step = sceneTime - lastSceneTime;
minStep = std::min(minStep, step);
maxStep = std::max(maxStep, step);
}
// Float interpolation weights can differ by a few nanoseconds at a boundary.
if (sceneTime + 1000 < lastSceneTime)
++backwards;
else if (sceneTime <= lastSceneTime + 1000)
++repeated;
}
lastSceneTime = sceneTime;
}
void clear_window() noexcept {
const auto last = lastSceneTime;
*this = {};
lastSceneTime = last;
}
};
} // namespace aurora::stereo
+723
View File
@@ -7,6 +7,7 @@
#include "gfx/texture.hpp"
#include "gx/shader_info.hpp"
#include "gx/pipeline.hpp"
#include "scene_camera.hpp"
#include "__gx.h"
#include <algorithm>
@@ -242,6 +243,631 @@ TEST_F(GXFifoTest, VrKeepsMatchedEndpointsWithDesktopInterpolationOff) {
EXPECT_EQ(build(40, 200, true).previous.size, 0u); // Readback split invalidates replay.
}
TEST_F(GXFifoTest, VrDiagnosticsExposeDistantCameraTurnRejectionDespiteMatchedIdentity) {
struct Reset {
~Reset() {
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::begin_frame_interpolation();
}
} reset;
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
const auto info = aurora::gx::build_shader_info({});
gxState().currentPnMtx = 0;
const auto build = [&](float yaw) {
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
const float c = std::cos(yaw), s = std::sin(yaw);
gxState().pnMtx[0].nrm = {{c, 0, s, 0}, {0, 1, 0, 0}, {-s, 0, c, 0}};
for (unsigned i = 0; i < 2; ++i) {
const float distance = i == 0 ? 1000.0f : 100000.0f;
// Two stationary objects, seen from one camera rotating by two degrees.
gxState().pnMtx[0].pos = {{c, 0, s, -s * distance}, {0, 1, 0, 0}, {-s, 0, c, -c * distance}};
aurora::gx::build_uniform(info, 0, {}, {100u + i, 100u + i, 7}, true);
}
aurora::gx::finalize_frame_interpolation();
};
build(0);
build(2.0f * 3.14159265f / 180.0f);
AuroraFrameInterpolationDiagnostics diagnostics{};
aurora::gx::get_frame_interpolation_diagnostics(diagnostics);
EXPECT_EQ(diagnostics.candidates, 2u);
EXPECT_EQ(diagnostics.matches, 2u);
EXPECT_EQ(diagnostics.preparedDraws, 1u);
EXPECT_EQ(diagnostics.rejectedDraws, 1u);
EXPECT_EQ(diagnostics.framesSealed, 2u); // VR-only frames must count too.
}
TEST_F(GXFifoTest, VrRetainsFastSpinningWheelEndpoints) {
struct Reset {
~Reset() {
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::begin_frame_interpolation();
}
} reset;
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
const auto info = aurora::gx::build_shader_info({});
gxState().currentPnMtx = 0;
const auto build = [&](float angle, float x) {
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
const float c = std::cos(angle), s = std::sin(angle);
gxState().pnMtx[0].pos = {{c, -s, 0, x}, {s, c, 0, 0}, {0, 0, 1, -50}};
gxState().pnMtx[0].nrm = {{c, -s, 0, 0}, {s, c, 0, 0}, {0, 0, 1, 0}};
const auto result = aurora::gx::build_uniform(info, 0, {}, {100, 42, 7}, true);
aurora::gx::finalize_frame_interpolation();
return result;
};
EXPECT_EQ(build(0, 10).previous.size, 0u);
const auto uniforms = build(2.0f * 3.14159265f / 3.0f, 30);
ASSERT_NE(uniforms.previous.size, 0u);
const auto& bytes = aurora::gfx::testing::uniform_allocation(uniforms.previous.offset);
aurora::Mat3x4<float> previousPosition{}, previousNormal{};
std::memcpy(static_cast<void*>(&previousPosition), bytes.data() + uniforms.replayLayout.positionOffset,
sizeof(previousPosition));
std::memcpy(static_cast<void*>(&previousNormal), bytes.data() + uniforms.replayLayout.normalOffset,
sizeof(previousNormal));
// Rejecting a >90-degree wheel spin used to copy the current position here too,
// leaving the entire wheel at 60 Hz even as the kart body moved smoothly.
EXPECT_FLOAT_EQ(previousPosition.m0.w(), 10);
EXPECT_FLOAT_EQ(previousPosition.m0.x(), 1);
EXPECT_FLOAT_EQ(previousNormal.m0.x(), 1);
AuroraFrameInterpolationDiagnostics diagnostics{};
aurora::gx::get_frame_interpolation_diagnostics(diagnostics);
EXPECT_EQ(diagnostics.matches, 1u);
EXPECT_EQ(diagnostics.preparedDraws, 1u);
EXPECT_EQ(diagnostics.rejectedDraws, 0u);
}
TEST_F(GXFifoTest, VrTextureAnimationRetainsSpatialHistoryWithStrictMeshIdentity) {
struct Reset {
~Reset() {
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::begin_frame_interpolation();
}
} reset;
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
const auto info = aurora::gx::build_shader_info({});
gxState().currentPnMtx = 0;
gxState().pnMtx[0].pos = {{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, -50}};
gxState().pnMtx[0].nrm = {{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
const auto begin = [&] {
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
};
const auto draw = [&](float x, aurora::HashType texture, aurora::HashType geometry = 123,
aurora::HashType topology = 0, aurora::HashType pipeline = 42) {
gxState().pnMtx[0].pos.m0[3] = x;
const aurora::gx::FrameInterpolationDrawIdentity identity{
.combined = texture + 1000, .pipeline = pipeline, .texture = texture,
.matrixTopology = topology, .geometry = geometry};
return aurora::gx::build_uniform(info, 0, {}, identity, true);
};
const auto previousX = [&](const auto& uniforms) {
const auto& bytes = aurora::gfx::testing::uniform_allocation(uniforms.previous.offset);
float x;
std::memcpy(&x, bytes.data() + uniforms.replayLayout.positionOffset + 3 * sizeof(float), sizeof(x));
return x;
};
begin();
draw(10, 1);
draw(100, 2);
draw(200, 3);
aurora::gx::finalize_frame_interpolation();
begin();
// The transparent sort reverses two animated instances. An unchanged exact
// match must also keep priority over the texture-independent mesh fallback.
const auto right = draw(110, 4);
const auto left = draw(20, 4);
const auto unchanged = draw(210, 3);
aurora::gx::finalize_frame_interpolation();
ASSERT_NE(right.previous.size, 0u);
ASSERT_NE(left.previous.size, 0u);
ASSERT_NE(unchanged.previous.size, 0u);
EXPECT_FLOAT_EQ(previousX(right), 100);
EXPECT_FLOAT_EQ(previousX(left), 10);
EXPECT_FLOAT_EQ(previousX(unchanged), 200);
AuroraFrameInterpolationDiagnostics diagnostics{};
aurora::gx::get_frame_interpolation_diagnostics(diagnostics);
EXPECT_EQ(diagnostics.preparedDraws, 3u);
EXPECT_EQ(diagnostics.rejectedDraws, 0u);
begin();
EXPECT_EQ(draw(120, 5, 456).previous.size, 0u); // Different mesh.
EXPECT_EQ(draw(120, 6, 0).previous.size, 0u); // Missing mesh identity.
EXPECT_EQ(draw(120, 7, 123, 9).previous.size, 0u); // Different palette topology.
EXPECT_EQ(draw(120, 8, 123, 0, 43).previous.size, 0u); // Different pipeline.
aurora::gx::finalize_frame_interpolation();
}
TEST(FrameInterpolationContract, FastRigidSpinsKeepAngularSpeedAndDiscontinuityGuards) {
const aurora::Mat3x4<float> previous{{1, 0, 0, 10}, {0, 1, 0, 0}, {0, 0, 1, 0}};
// Include rotations whose quaternion representation needs hemisphere correction.
for (float degrees : {120.0f, 170.0f, -120.0f, 240.0f}) {
const float angle = degrees * 3.14159265f / 180.0f;
const float c = std::cos(angle), s = std::sin(angle);
const aurora::Mat3x4<float> current{{c, -s, 0, 30}, {s, c, 0, 0}, {0, 0, 1, 0}};
aurora::Mat3x4<float> output{};
EXPECT_FALSE(aurora::gx::interpolate_transform(previous, current, 0.5f, output)); // Anchor cut guard.
for (float weight : {0.0f, 1.0f / 3, 2.0f / 3, 1.0f}) {
ASSERT_TRUE(aurora::gx::interpolate_draw_transform(previous, current, weight, output));
const float expectedAngle = (degrees > 180 ? degrees - 360 : degrees) * 3.14159265f / 180.0f * weight;
EXPECT_NEAR(output.m0.x(), std::cos(expectedAngle), 1e-5f);
EXPECT_NEAR(output.m1.x(), std::sin(expectedAngle), 1e-5f);
EXPECT_NEAR(output.m0.w(), 10 + 20 * weight, 1e-5f);
EXPECT_NEAR(output.m0.x() * output.m0.x() + output.m1.x() * output.m1.x(), 1, 1e-5f);
}
}
aurora::Mat3x4<float> invalid = previous, output{};
invalid.m0[3] = 2010;
EXPECT_FALSE(aurora::gx::interpolate_draw_transform(previous, invalid, 0.5f, output));
EXPECT_FLOAT_EQ(output.m0.w(), 2010);
invalid = previous;
invalid.m0[0] = 0; // Singular axis.
EXPECT_FALSE(aurora::gx::interpolate_draw_transform(previous, invalid, 0.5f, output));
invalid = previous;
invalid.m0[3] = std::numeric_limits<float>::quiet_NaN();
EXPECT_FALSE(aurora::gx::interpolate_draw_transform(previous, invalid, 0.5f, output));
}
TEST(FrameInterpolationContract, ShearedRigidDrawsMoveWithTheSkinnedMeshOnTheirBone) {
// Lakitu::Movement::UpdateScale sways MKW's Lakitu by tilting his Y axis: the model
// matrix gets a Y column of (A cos p, 1, A sin p). His goggles are rigid on the face
// bone and his head is skinned to that same bone. The rigid path rejected the shear,
// holding the goggles at the game frame while the head moved on, so they sank in.
const auto swaying = [](float amplitude, float phase, float lift) {
const float c = std::cos(0.698f), s = std::sin(0.698f); // the face bone's roll
const float x = amplitude * std::cos(phase), z = amplitude * std::sin(phase);
// [[1, x, 0], [0, 1, 0], [0, z, 1]] * Rz, then placed in front of the camera.
return aurora::Mat3x4<float>{{c + x * s, -s + x * c, 0, 10}, {s, c, 0, 50 + lift}, {z * s, z * c, 1, -300}};
};
const auto previous = swaying(0.25f, 0.3f, 0), current = swaying(0.3f, 0.5f, 3);
const auto apply = [](const aurora::Mat3x4<float>& matrix, const std::array<float, 3>& point) {
const aurora::Vec4<float>* rows[] = {&matrix.m0, &matrix.m1, &matrix.m2};
std::array<float, 3> result{};
for (size_t row = 0; row < 3; ++row)
result[row] = (*rows[row])[0] * point[0] + (*rows[row])[1] * point[1] + (*rows[row])[2] * point[2] +
(*rows[row])[3];
return result;
};
const std::array<float, 3> goggleCorner{30, -20, 10};
for (float weight : {0.0f, 0.25f, 0.5f, 0.75f, 1.0f}) {
aurora::Mat3x4<float> head{}, goggles{};
ASSERT_TRUE(aurora::gx::interpolate_indexed_transform(previous, current, weight, head));
ASSERT_TRUE(aurora::gx::interpolate_draw_transform(previous, current, weight, goggles));
const auto onHead = apply(head, goggleCorner), onGoggles = apply(goggles, goggleCorner);
for (size_t axis = 0; axis < 3; ++axis) EXPECT_NEAR(onGoggles[axis], onHead[axis], 0.05f) << weight;
}
// Camera and seat anchors are rigid; a sheared one still counts as a cut.
aurora::Mat3x4<float> anchor{};
EXPECT_FALSE(aurora::gx::interpolate_transform(previous, current, 0.5f, anchor));
}
TEST_F(GXFifoTest, VrCameraRebaseKeepsFarInstancesAndHeldParticlesContinuous) {
using aurora::Mat3x4;
using aurora::gfx::stereo_replay::compose_affine;
struct Reset {
~Reset() {
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::begin_frame_interpolation();
}
} reset;
aurora::gx::set_frame_interpolation_fps(120); // Desktop and VR enabled together.
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
const auto info = aurora::gx::build_shader_info({});
gxState().currentPnMtx = 0;
const Mat3x4<float> identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
const float angle = 0.08f, c = std::cos(angle), s = std::sin(angle);
const Mat3x4<float> currentView{{c, 0, s, 0}, {0, 1, 0, 0}, {-s, 0, c, 0}};
aurora::stereo::SceneCameraMotion motion;
ASSERT_TRUE(motion.prepare(identity, currentView, identity, identity));
const auto draw = [&](const Mat3x4<float>& view, float x, bool particle = false) {
gxState().vtxDesc[GX_VA_POS] = particle ? GX_DIRECT : GX_INDEX16;
gxState().pnMtx[0].pos = particle ? identity : compose_affine(view, {{1, 0, 0, x}, {0, 1, 0, 0}, {0, 0, 1, -100000}});
if (particle) gxState().pnMtx[0].pos.m0[1] = -0.0f;
gxState().pnMtx[0].nrm = particle ? identity : view;
return aurora::gx::build_uniform(info, 0, {}, particle ? aurora::gx::FrameInterpolationDrawIdentity{200, 20, 1}
: aurora::gx::FrameInterpolationDrawIdentity{100, 10, 1}, true);
};
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
draw(identity, -2000);
draw(identity, 2000);
draw(identity, 0, true);
aurora::gx::finalize_frame_interpolation();
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
// All far instances move over the old 1500-unit gate just from camera yaw.
// Submission order also changes, and the right instance moves 30 world units.
const auto right = draw(currentView, 2030);
const auto left = draw(currentView, -2000);
const auto particle = draw(currentView, 0, true);
aurora::gx::set_frame_interpolation_view_rebase(&motion.currentFromPrevious, &motion.previousFromCurrent);
aurora::gx::finalize_frame_interpolation();
const auto read = [&](const auto& uniform, aurora::gfx::Range range) {
Mat3x4<float> matrix;
const auto& bytes = aurora::gfx::testing::uniform_allocation(range.offset);
std::memcpy(static_cast<void*>(&matrix), bytes.data() + uniform.replayLayout.positionOffset, sizeof(matrix));
return matrix;
};
ASSERT_NE(right.previous.size, 0u);
ASSERT_NE(left.previous.size, 0u);
ASSERT_NE(particle.previous.size, 0u);
const auto previousRight = read(right, right.previous);
const auto expectedRight = compose_affine(currentView, {{1, 0, 0, 2000}, {0, 1, 0, 0}, {0, 0, 1, -100000}});
EXPECT_NEAR(previousRight.m0.w(), expectedRight.m0.w(), 0.01f);
EXPECT_NEAR(previousRight.m2.w(), expectedRight.m2.w(), 0.01f);
EXPECT_NEAR(read(left, left.previous).m0.w(), read(left, left.current).m0.w(), 0.01f);
EXPECT_FLOAT_EQ(read(particle, particle.previous).m0.x(), 1);
EXPECT_FLOAT_EQ(read(particle, particle.previous).m0.w(), 0); // No camera applied twice to baked vertices.
// Desktop retains its original guarded camera-space behavior.
ASSERT_NE(right.interpolated[0].size, 0u);
EXPECT_NEAR(read(right, right.interpolated[0]).m0.w(), read(right, right.current).m0.w(), 0.01f);
AuroraFrameInterpolationDiagnostics diagnostics{};
aurora::gx::get_frame_interpolation_diagnostics(diagnostics);
EXPECT_EQ(diagnostics.matches, 3u);
EXPECT_EQ(diagnostics.preparedDraws, 3u);
EXPECT_EQ(diagnostics.rejectedDraws, 0u);
}
TEST_F(GXFifoTest, VrParticleCentersFollowMotionAcrossSortChangesAndCameraTurns) {
using aurora::Mat3x4;
using aurora::gx::offset_transform_origin;
using aurora::gfx::stereo_replay::compose_affine;
struct Reset {
~Reset() {
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::begin_frame_interpolation();
}
} reset;
aurora::gx::set_frame_interpolation_fps(120);
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
const auto info = aurora::gx::build_shader_info({});
const Mat3x4<float> identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
const float angle = 0.08f, c = std::cos(angle), s = std::sin(angle);
const Mat3x4<float> view{{c, 0, s, 0}, {0, 1, 0, 0}, {-s, 0, c, 0}};
aurora::stereo::SceneCameraMotion motion;
ASSERT_TRUE(motion.prepare(identity, view, identity, identity));
gxState().vtxDesc[GX_VA_POS] = GX_DIRECT;
gxState().pnMtx[0].pos = gxState().pnMtx[0].nrm = identity;
const auto center = [&](const Mat3x4<float>& camera, float x) {
const auto matrix = offset_transform_origin(camera, {x, 0, -5000});
return std::array<float, 3>{matrix.m0.w(), matrix.m1.w(), matrix.m2.w()};
};
const auto record = [&](const Mat3x4<float>& camera, float x, uint64_t geometry) {
return aurora::gx::build_uniform(info, 0, {}, {geometry, 7, 9, 0, geometry}, true, 1,
{center(camera, x), true});
};
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
record(identity, -600, 101);
record(identity, 600, 102);
aurora::gx::finalize_frame_interpolation();
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
// Vertex bytes change and transparent submission order reverses. Both draw
// matrices are identity, so matrix-only matching cannot identify the centers.
const auto right = record(view, 630, 201);
const auto left = record(view, -570, 202);
const auto newborn = record(view, 10000, 203);
aurora::gx::set_frame_interpolation_view_rebase(&motion.currentFromPrevious, &motion.previousFromCurrent);
aurora::gx::finalize_frame_interpolation();
const auto readCenter = [&](const auto& uniform, aurora::gfx::Range range) {
Mat3x4<float> matrix;
const auto& bytes = aurora::gfx::testing::uniform_allocation(range.offset);
std::memcpy(&matrix, bytes.data() + uniform.replayLayout.positionOffset, sizeof(matrix));
EXPECT_FLOAT_EQ(matrix.m0.x(), 1); // Keep the current billboard's orientation.
EXPECT_FLOAT_EQ(matrix.m2.x(), 0);
return offset_transform_origin(matrix, uniform.replayLayout.vertexMotion.center);
};
for (const auto& [uniform, x] : {std::pair{right, 600.f}, std::pair{left, -600.f}}) {
ASSERT_NE(uniform.previous.size, 0u);
const auto previous = readCenter(uniform, uniform.previous);
EXPECT_NEAR(previous.m0.w(), center(view, x)[0], 0.002f);
EXPECT_NEAR(previous.m2.w(), center(view, x)[2], 0.002f);
const auto current = readCenter(uniform, uniform.current);
Mat3x4<float> half;
ASSERT_TRUE(aurora::gx::interpolate_draw_transform(previous, current, 0.5f, half));
EXPECT_NEAR(half.m0.w(), center(view, x + 15)[0], 0.002f);
// Simultaneous desktop interpolation uses its original camera endpoint.
const auto desktop = readCenter(uniform, uniform.interpolated[0]);
EXPECT_NEAR(desktop.m0.w(), (center(identity, x)[0] + center(view, x + 30)[0]) * 0.5f, 0.002f);
}
EXPECT_NEAR(readCenter(newborn, newborn.previous).m0.w(), center(view, 10000)[0], 0.002f);
AuroraFrameInterpolationDiagnostics diagnostics{};
aurora::gx::get_frame_interpolation_diagnostics(diagnostics);
EXPECT_EQ(diagnostics.vertexMotionDraws, 2u);
}
namespace {
using Vec3f = std::array<float, 3>;
// One CPU-built particle quad the way nw4r::ef submits it: camera-space corners under
// an identity position matrix, all quads of an emitter sharing pipeline and texture.
aurora::gx::UniformRanges record_particle_quad(const aurora::gx::ShaderInfo& info, uint64_t geometry,
const Vec3f& center, const Vec3f& edge0, const Vec3f& edge1) {
return aurora::gx::build_uniform(info, 0, {}, {geometry, 7, 9, 0, geometry}, true, 1, {center, true},
{edge0, edge1});
}
// The centre a staged uniform draws the quad at.
Vec3f drawn_center(const aurora::gx::UniformRanges& uniform, aurora::gfx::Range range) {
aurora::Mat3x4<float> matrix;
const auto& bytes = aurora::gfx::testing::uniform_allocation(range.offset);
std::memcpy(static_cast<void*>(&matrix), bytes.data() + uniform.replayLayout.positionOffset, sizeof(matrix));
matrix = aurora::gx::offset_transform_origin(matrix, uniform.replayLayout.vertexMotion.center);
return {matrix.m0.w(), matrix.m1.w(), matrix.m2.w()};
}
Vec3f transform_point(const aurora::Mat3x4<float>& matrix, const Vec3f& point) {
return {matrix.m0[0] * point[0] + matrix.m0[1] * point[1] + matrix.m0[2] * point[2] + matrix.m0[3],
matrix.m1[0] * point[0] + matrix.m1[1] * point[1] + matrix.m1[2] * point[2] + matrix.m1[3],
matrix.m2[0] * point[0] + matrix.m2[1] * point[1] + matrix.m2[2] * point[2] + matrix.m2[3]};
}
bool same_point(const Vec3f& a, const Vec3f& b, float tolerance = 0.05f) {
return std::abs(a[0] - b[0]) <= tolerance && std::abs(a[1] - b[1]) <= tolerance &&
std::abs(a[2] - b[2]) <= tolerance;
}
struct ParticleInterpolationReset {
~ParticleInterpolationReset() {
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::begin_frame_interpolation();
}
};
} // namespace
TEST_F(GXFifoTest, VrParticleQuadsPairByShapeWhereNearestCentresSwap) {
ParticleInterpolationReset reset;
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
const auto info = aurora::gx::build_shader_info({});
const aurora::Mat3x4<float> identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
gxState().currentPnMtx = 0;
gxState().vtxDesc[GX_VA_POS] = GX_DIRECT;
gxState().pnMtx[0].pos = gxState().pnMtx[0].nrm = identity;
// The camera flies 100 units forward and carries two streaks, each moving 60 along its
// own length. The one lying across the view ends nearer the other's old centre, so
// pairing centres alone swaps them and both sweep sideways.
const aurora::Mat3x4<float> forward{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 100}};
aurora::stereo::SceneCameraMotion motion;
ASSERT_TRUE(motion.prepare(identity, forward, identity, identity));
const Vec3f across{300, 0, 0}, upright{0, 300, 0}, thinX{12, 0, 0}, thinY{0, 12, 0};
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
record_particle_quad(info, 1, {-20, 0, -500}, across, thinY);
record_particle_quad(info, 2, {20, 0, -500}, upright, thinX);
aurora::gx::finalize_frame_interpolation();
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
const auto lying = record_particle_quad(info, 3, {40, 0, -500}, across, thinY);
const auto standing = record_particle_quad(info, 4, {20, 60, -500}, upright, thinX);
aurora::gx::set_frame_interpolation_view_rebase(&motion.currentFromPrevious, &motion.previousFromCurrent);
aurora::gx::finalize_frame_interpolation();
ASSERT_NE(lying.previous.size, 0u);
ASSERT_NE(standing.previous.size, 0u);
// Each starts from its own old place, as the new camera position sees it.
EXPECT_TRUE(same_point(drawn_center(lying, lying.previous), {-20, 0, -400}));
EXPECT_TRUE(same_point(drawn_center(standing, standing.previous), {20, 0, -400}));
AuroraFrameInterpolationDiagnostics diagnostics{};
aurora::gx::get_frame_interpolation_diagnostics(diagnostics);
EXPECT_EQ(diagnostics.vertexMotionDraws, 2u);
EXPECT_EQ(diagnostics.vertexMotionHeld, 0u);
}
TEST_F(GXFifoTest, VrParticleQuadBornElsewhereDoesNotSweepFromOneThatDied) {
ParticleInterpolationReset reset;
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
const auto info = aurora::gx::build_shader_info({});
const aurora::Mat3x4<float> identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
gxState().currentPnMtx = 0;
gxState().vtxDesc[GX_VA_POS] = GX_DIRECT;
gxState().pnMtx[0].pos = gxState().pnMtx[0].nrm = identity;
const aurora::Mat3x4<float> forward{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 100}};
aurora::stereo::SceneCameraMotion motion;
ASSERT_TRUE(motion.prepare(identity, forward, identity, identity));
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
record_particle_quad(info, 1, {-100, 0, -500}, {300, 0, 0}, {0, 12, 0});
aurora::gx::finalize_frame_interpolation();
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
// That streak died; another one starts elsewhere. One quad on each side of the frame
// is no evidence that they are the same particle.
const auto newborn = record_particle_quad(info, 2, {100, 0, -500}, {0, 300, 0}, {12, 0, 0});
aurora::gx::set_frame_interpolation_view_rebase(&motion.currentFromPrevious, &motion.previousFromCurrent);
aurora::gx::finalize_frame_interpolation();
ASSERT_NE(newborn.previous.size, 0u);
EXPECT_TRUE(same_point(drawn_center(newborn, newborn.previous), {100, 0, -500}));
AuroraFrameInterpolationDiagnostics diagnostics{};
aurora::gx::get_frame_interpolation_diagnostics(diagnostics);
EXPECT_EQ(diagnostics.vertexMotionDraws, 0u);
EXPECT_EQ(diagnostics.vertexMotionHeld, 1u);
}
TEST_F(GXFifoTest, VrSpeedLineEmitterKeepsEachStreakOnItsOwnPath) {
// A deterministic stand-in for the boost speed lines (rk_koukasen in RKRace.breff):
// 12x300 streaks, each drawn as two crossed quads, two born per frame on a 90-unit
// ring, six-frame life, about 50 units per frame outwards and back, all carried by a
// camera that flies 100 units per frame while turning. Pairing nearest centres swapped
// about half of them and swept every newborn in from a streak that had just died.
ParticleInterpolationReset reset;
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
const auto info = aurora::gx::build_shader_info({});
const aurora::Mat3x4<float> identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
gxState().currentPnMtx = 0;
gxState().vtxDesc[GX_VA_POS] = GX_DIRECT;
gxState().pnMtx[0].pos = gxState().pnMtx[0].nrm = identity;
struct Streak {
uint32_t id, birth;
Vec3f origin, velocity;
};
struct Quad {
uint32_t streak, side;
Vec3f center, edge0, edge1;
};
uint32_t seed = 0x5eed1234u;
const auto random = [&seed] {
seed = seed * 1664525u + 1013904223u;
return static_cast<float>(seed >> 8) * (1.0f / 16777216.0f);
};
const auto quads_of = [](const Streak& streak, uint32_t frame) {
const float age = static_cast<float>(frame - streak.birth);
const auto& v = streak.velocity;
const float speed = std::sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]);
const Vec3f axis{v[0] / speed, v[1] / speed, v[2] / speed};
// Two sides across the axis: the crossed planes of one streak.
const float across = std::sqrt(axis[0] * axis[0] + axis[2] * axis[2]);
const Vec3f side0{axis[2] / across, 0, -axis[0] / across};
const Vec3f side1{axis[1] * side0[2] - axis[2] * side0[1], axis[2] * side0[0] - axis[0] * side0[2],
axis[0] * side0[1] - axis[1] * side0[0]};
Vec3f center{};
for (size_t i = 0; i < 3; ++i) center[i] = streak.origin[i] + v[i] * age - axis[i] * 150;
std::array<Quad, 2> quads{};
for (uint32_t side = 0; side < 2; ++side) {
const auto& s = side == 0 ? side0 : side1;
quads[side] = {streak.id, side, center, {axis[0] * 300, axis[1] * 300, axis[2] * 300},
{s[0] * 12, s[1] * 12, s[2] * 12}};
}
return quads;
};
std::vector<Streak> streaks;
std::vector<Quad> previousQuads;
aurora::Mat3x4<float> previousView = identity;
Vec3f cameraPosition{};
float yaw = 0;
uint32_t nextId = 0, survivors = 0, moved = 0, held = 0, wrong = 0, newborns = 0, newbornsMoved = 0;
for (uint32_t frame = 0; frame < 90; ++frame) {
yaw += 0.01f;
const float c = std::cos(yaw), s = std::sin(yaw);
cameraPosition = {cameraPosition[0] - s * 100, 0, cameraPosition[2] - c * 100};
// Camera looks down -Z, rotated by yaw about +Y; the view is its inverse.
const aurora::Mat3x4<float> view{
{c, 0, -s, -(c * cameraPosition[0] - s * cameraPosition[2])},
{0, 1, 0, 0},
{s, 0, c, -(s * cameraPosition[0] + c * cameraPosition[2])}};
aurora::stereo::SceneCameraMotion motion;
const bool rebase = frame != 0 && motion.prepare(previousView, view, identity, identity);
ASSERT_TRUE(frame == 0 || rebase);
std::erase_if(streaks, [frame](const Streak& streak) { return frame - streak.birth >= 6; });
for (int born = 0; born < 2; ++born) {
const float angle = random() * 6.2831853f;
const float speed = 1.0f + (random() - 0.5f) * 0.46f;
streaks.push_back({nextId++, frame, {90 * std::cos(angle), 90 * std::sin(angle), -300},
{40 * speed * std::cos(angle), 40 * speed * std::sin(angle) + 10 * speed, 30 * speed}});
}
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
std::vector<Quad> quads;
std::vector<aurora::gx::UniformRanges> uniforms;
for (const auto& streak : streaks) {
for (const auto& quad : quads_of(streak, frame)) {
quads.push_back(quad);
uniforms.push_back(record_particle_quad(info, 1000 + quads.size() + frame * 100, quad.center,
quad.edge0, quad.edge1));
}
}
if (rebase)
aurora::gx::set_frame_interpolation_view_rebase(&motion.currentFromPrevious, &motion.previousFromCurrent);
aurora::gx::finalize_frame_interpolation();
if (rebase) {
for (size_t index = 0; index < quads.size(); ++index) {
const auto& quad = quads[index];
if (uniforms[index].previous.size == 0) continue;
const auto start = drawn_center(uniforms[index], uniforms[index].previous);
const auto before = std::find_if(previousQuads.begin(), previousQuads.end(), [&](const Quad& old) {
return old.streak == quad.streak && old.side == quad.side;
});
const bool heldHere = same_point(start, quad.center) ||
same_point(start, transform_point(motion.currentFromPrevious, quad.center));
if (before == previousQuads.end()) {
++newborns;
newbornsMoved += !heldHere;
continue;
}
++survivors;
if (heldHere) {
++held;
} else if (same_point(start, transform_point(motion.currentFromPrevious, before->center))) {
++moved;
} else {
++wrong;
}
}
}
previousQuads = std::move(quads);
previousView = view;
}
ASSERT_GT(survivors, 800u);
// Most streak quads keep moving along their own path...
EXPECT_GT(moved, survivors / 2);
// ...and almost none borrows another streak's (it was about half, plus every newborn).
EXPECT_LE(wrong * 50, survivors);
EXPECT_LE(newbornsMoved * 20, newborns);
std::printf("speed lines: %u survivor quads: %u moved, %u held, %u wrong; %u of %u newborn quads moved\n",
survivors, moved, held, wrong, newbornsMoved, newborns);
}
TEST_F(GXFifoTest, VrCyclicRigidMotionDoesNotBlendBackwardsAcrossReset) {
struct Reset {
~Reset() {
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
aurora::gx::begin_frame_interpolation();
}
} reset;
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
const auto info = aurora::gx::build_shader_info({});
const aurora::Mat3x4<float> identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
gxState().vtxDesc[GX_VA_POS] = GX_INDEX16;
auto previousView = identity;
const auto record = [&](float phase, float yaw) {
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::reset_uniform_allocations();
const aurora::Mat3x4<float> view{{std::cos(yaw), 0, std::sin(yaw), 0}, {0, 1, 0, 0},
{-std::sin(yaw), 0, std::cos(yaw), 0}};
aurora::stereo::SceneCameraMotion camera;
EXPECT_TRUE(camera.prepare(previousView, view, identity, identity));
gxState().pnMtx[0].pos = aurora::gx::offset_transform_origin(view, {phase, 0, -5000});
gxState().pnMtx[0].nrm = view;
const auto result = aurora::gx::build_uniform(info, 0, {}, {11, 1, 2, 0, 11}, true);
aurora::gx::set_frame_interpolation_view_rebase(&camera.currentFromPrevious, &camera.previousFromCurrent);
aurora::gx::finalize_frame_interpolation();
previousView = view;
return result;
};
record(16, 0);
record(17, 0.02f);
record(18, 0.04f);
record(19, 0.06f);
AuroraFrameInterpolationDiagnostics before{}, after{};
aurora::gx::get_frame_interpolation_diagnostics(before);
const auto resetFrame = record(0, 0.08f);
aurora::gx::get_frame_interpolation_diagnostics(after);
EXPECT_EQ(after.animationWrapCuts, before.animationWrapCuts + 1);
const auto& current = aurora::gfx::testing::uniform_allocation(resetFrame.current.offset);
const auto& previous = aurora::gfx::testing::uniform_allocation(resetFrame.previous.offset);
EXPECT_EQ(current, previous); // Hold the new phase instead of reverse sweeping.
record(1, 0.10f);
record(0, 0.12f); // A normal same-speed direction change must still interpolate.
aurora::gx::get_frame_interpolation_diagnostics(after);
EXPECT_EQ(after.animationWrapCuts, before.animationWrapCuts + 1);
EXPECT_EQ(after.preparedDraws, 1u);
}
TEST(FrameInterpolationContract, RequiresStablePerspectiveDrawSequence) {
const auto resetInterpolation = [] {
aurora::gx::set_frame_interpolation_fps(0);
@@ -2269,6 +2895,103 @@ TEST_F(GXFifoTest, MergedDrawOffsetsCachedTopologyWithoutJoiningPrimitives) {
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{3, 4, 5}));
}
TEST_F(GXFifoTest, VrDirectParticleQuadsRetainSeparateCentersInFifoAndRawDraws) {
struct Reset {
~Reset() {
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
aurora::gx::begin_frame_interpolation();
}
} reset;
aurora::gx::detail::g_stereoFrameInterpolation.store(true);
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::use_real_vertex_format_helpers(true);
aurora::gfx::testing::use_draw_command_tracking(true);
gxState().projType = GX_PERSPECTIVE;
gxState().vtxDesc[GX_VA_POS] = GX_DIRECT;
gxState().vtxFmts[GX_VTXFMT0].attrs[GX_VA_POS] = {GX_POS_XYZ, GX_F32, 0};
gxState().pnMtx[0].pos = gxState().pnMtx[0].nrm = {{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
gxState().stateDirty = true;
const auto makeVertices = [](float x) {
std::vector<uint8_t> vertices;
for (const auto& point : {std::array{x - 5, -5.f, -200.f}, std::array{x + 5, -5.f, -200.f},
std::array{x + 5, 5.f, -200.f}, std::array{x - 5, 5.f, -200.f}}) {
for (float component : point) {
uint32_t bits;
std::memcpy(&bits, &component, sizeof(bits));
for (int shift : {24, 16, 8, 0}) vertices.push_back(static_cast<uint8_t>(bits >> shift));
}
}
return vertices;
};
std::vector<uint8_t> commands;
for (float x : {20.f, 100.f}) {
commands.insert(commands.end(), {static_cast<uint8_t>(GX_QUADS), 0, 4});
const auto vertices = makeVertices(x);
commands.insert(commands.end(), vertices.begin(), vertices.end());
}
decode_fifo(commands);
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
auto* draw = aurora::gfx::get_last_draw_command<aurora::gx::DrawData>();
ASSERT_NE(draw, nullptr);
EXPECT_TRUE(draw->uniformReplayLayout.vertexMotion.enabled);
EXPECT_EQ(draw->uniformReplayLayout.vertexMotion.center, (std::array{100.f, 0.f, -200.f}));
const auto raw = makeVertices(300);
ASSERT_TRUE(aurora::gx::fifo::submit_raw_draw(GX_QUADS, GX_VTXFMT0, raw.data(), 4, raw.size()));
draw = aurora::gfx::get_last_draw_command<aurora::gx::DrawData>();
ASSERT_NE(draw, nullptr);
EXPECT_TRUE(draw->uniformReplayLayout.vertexMotion.enabled);
EXPECT_EQ(draw->uniformReplayLayout.vertexMotion.center, (std::array{300.f, 0.f, -200.f}));
// Turning VR interpolation off restores ordinary batching.
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
gxState().stateDirty = true;
decode_fifo(commands);
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 1u);
}
TEST_F(GXFifoTest, InterpolationOffRecordsNothingForParticlesOrMatching) {
// The Quest ships with VR interpolation off. Those frames must not pay for any
// of it: no draw is recorded, no particle quad is tracked, nothing is matched.
struct Reset {
~Reset() { aurora::gx::begin_frame_interpolation(); }
} reset;
aurora::gx::detail::g_stereoFrameInterpolation.store(false);
aurora::gx::set_frame_interpolation_fps(0);
aurora::gx::begin_frame_interpolation();
aurora::gfx::testing::use_real_vertex_format_helpers(true);
aurora::gfx::testing::use_draw_command_tracking(true);
gxState().projType = GX_PERSPECTIVE;
gxState().vtxDesc[GX_VA_POS] = GX_DIRECT;
gxState().vtxFmts[GX_VTXFMT0].attrs[GX_VA_POS] = {GX_POS_XYZ, GX_F32, 0};
gxState().pnMtx[0].pos = gxState().pnMtx[0].nrm = {{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
gxState().stateDirty = true;
std::vector<uint8_t> commands;
for (float x : {20.f, 100.f, 180.f}) {
commands.insert(commands.end(), {static_cast<uint8_t>(GX_QUADS), 0, 4});
for (const auto& point : {std::array{x - 5, -5.f, -200.f}, std::array{x + 5, -5.f, -200.f},
std::array{x + 5, 5.f, -200.f}, std::array{x - 5, 5.f, -200.f}}) {
for (float component : point) {
uint32_t bits;
std::memcpy(&bits, &component, sizeof(bits));
for (int shift : {24, 16, 8, 0}) commands.push_back(static_cast<uint8_t>(bits >> shift));
}
}
}
decode_fifo(commands);
aurora::gx::finalize_frame_interpolation();
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 2u); // Ordinary batching.
auto* draw = aurora::gfx::get_last_draw_command<aurora::gx::DrawData>();
ASSERT_NE(draw, nullptr);
EXPECT_FALSE(draw->uniformReplayLayout.vertexMotion.enabled);
AuroraFrameInterpolationDiagnostics diagnostics{};
aurora::gx::get_frame_interpolation_diagnostics(diagnostics);
EXPECT_EQ(diagnostics.candidates, 0u);
EXPECT_EQ(diagnostics.matches, 0u);
EXPECT_EQ(diagnostics.preparedDraws, 0u);
EXPECT_EQ(diagnostics.vertexMotionDraws, 0u);
EXPECT_EQ(diagnostics.vertexMotionHeld, 0u);
EXPECT_FALSE(aurora::gx::has_interpolated_frame());
}
TEST_F(GXFifoTest, OrthographicQuadRecordsScreenRectForVrFurniture) {
// MKW draws its split-screen partition with the partition_line layout: a
// one-pixel picture pane sampling a pattern texture in a full-display
+49 -12
View File
@@ -1,4 +1,4 @@
// Optional GPU smoke test: a 60 Hz GX producer with an independent 90 Hz
// Optional GPU smoke test: a 60 Hz GX producer with an independent headset-rate
// compositor. Exercises the real frame worker, eye replay and submission sink.
#include <aurora/aurora.h>
#include <aurora/gfx.h>
@@ -9,9 +9,11 @@
#include <algorithm>
#include <atomic>
#include <chrono>
#include <cmath>
#include <condition_variable>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <filesystem>
#include <mutex>
#include <thread>
@@ -56,6 +58,7 @@ static bool stop = false;
static uint64_t completed = 0;
static std::atomic_uint32_t submitted{0};
static uint64_t wakeLateness = 0, submitTime = 0, skippedTicks = 0, compositorFrames = 0;
static uint64_t headsetHz = 90, predictionLeadNanos = 40'000'000;
static bool Provide(uint32_t, AuroraStereoFrame* output, void*) {
std::lock_guard lock(packetMutex);
@@ -73,7 +76,7 @@ static void Submitted(const aurora::stereo::SinkFrame& frame, void*) noexcept {
packetCv.notify_all();
}
static void Log(AuroraLogLevel level, const char* module, const char* message, unsigned int length) {
if (level >= LOG_WARNING)
if (level >= LOG_WARNING || std::strstr(message, "[vr-motion]") != nullptr)
std::fprintf(stderr, "%s: %.*s\n", module, static_cast<int>(length), message);
}
@@ -81,8 +84,12 @@ int main(int argc, char** argv) {
// Extra distinct draws expose CPU uniform/replay costs that a single triangle
// cannot exercise. Keep the eye targets small to isolate that regression.
const unsigned drawCount = argc > 1 ? std::max(1, std::atoi(argv[1])) : 1;
const bool movingCamera = argc > 6 && std::atoi(argv[6]) != 0;
const bool particles = argc > 7 && std::atoi(argv[7]) != 0;
const int mode = argc < 3 ? 1 : std::clamp(std::atoi(argv[2]), 0, 2);
const bool indexed = argc > 3 && std::atoi(argv[3]) != 0;
const bool indexed = !particles && argc > 3 && std::atoi(argv[3]) != 0;
headsetHz = argc > 4 ? std::clamp(std::atoi(argv[4]), 60, 120) : 90;
predictionLeadNanos = argc > 5 ? std::clamp(std::atoi(argv[5]), 0, 100) * 1'000'000ull : 40'000'000;
std::filesystem::create_directories("stereo-smoke-cache");
AuroraConfig config{};
config.appName = "Aurora VR interpolation smoke";
@@ -95,11 +102,12 @@ int main(int argc, char** argv) {
config.windowPosX = -30000;
config.windowPosY = -30000;
config.logCallback = Log;
config.logLevel = LOG_WARNING;
config.logLevel = LOG_INFO;
config.xrInterop = true;
aurora_initialize(argc, argv, &config);
aurora_set_frame_interpolation_fps(0);
aurora_set_stereo_frame_interpolation(mode != 0);
aurora_set_stereo_motion_logging(true);
aurora_set_stereo_frame_provider(Provide, nullptr);
aurora::stereo::set_sink(Encode, Submitted, nullptr);
@@ -107,7 +115,7 @@ int main(int argc, char** argv) {
const auto start = Clock::now();
uint64_t slot = 1;
for (uint64_t token = 1;; ++token) {
const auto deadline = start + std::chrono::nanoseconds(slot * 1'000'000'000 / 90);
const auto deadline = start + std::chrono::nanoseconds(slot * 1'000'000'000 / headsetHz);
WaitUntil(deadline);
const auto woke = Clock::now();
wakeLateness +=
@@ -119,11 +127,11 @@ int main(int argc, char** argv) {
packet = {};
packet.frameToken = token;
packet.contentTag = 42;
// Wake to render the next display tick, as xrWaitFrame does, rather
// than announcing an image whose display deadline has already passed.
// A runtime can predict multiple frames ahead. This must not exhaust
// scene history while the producer and compositor maintain their rates.
packet.displayTimeNanos =
std::chrono::duration_cast<std::chrono::nanoseconds>(deadline.time_since_epoch()).count() +
1'000'000'000 / 90;
predictionLeadNanos;
for (auto& eye : packet.eyes) {
eye.width = 160;
eye.height = 120;
@@ -146,7 +154,7 @@ int main(int argc, char** argv) {
const auto elapsed = std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - start).count();
submitTime += std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - woke).count();
++compositorFrames;
const auto nextSlot = std::max(slot + 1, static_cast<uint64_t>(elapsed) * 90 / 1'000'000'000);
const auto nextSlot = std::max(slot + 1, static_cast<uint64_t>(elapsed) * headsetHz / 1'000'000'000);
skippedTicks += nextSlot - slot - 1;
slot = nextSlot;
}
@@ -178,6 +186,19 @@ int main(int argc, char** argv) {
std::chrono::duration_cast<std::chrono::nanoseconds>(boundary.time_since_epoch()).count(), 16'666'667);
Mtx44 projection{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, -1, -1}, {0, 0, -1, 0}};
Mtx transform{{1, 0, 0, static_cast<float>(frame % 60) * 0.01f}, {0, 1, 0, 0}, {0, 0, 1, -3}};
const float yaw = movingCamera ? std::sin(static_cast<float>(frame) * 0.08f) * 0.5f : 0;
Mtx sceneView{{std::cos(yaw), 0, std::sin(yaw), 0}, {0, 1, 0, 0}, {-std::sin(yaw), 0, std::cos(yaw), 0}};
if (movingCamera) {
transform[2][3] = -100000;
aurora_set_stereo_scene_view(&sceneView[0][0]);
}
const auto applyView = [&](Mtx viewed) {
for (unsigned row = 0; row < 3; ++row)
for (unsigned col = 0; col < 4; ++col)
viewed[row][col] = (col == 3 ? sceneView[row][3] : 0.f) +
sceneView[row][0] * transform[0][col] + sceneView[row][1] * transform[1][col] +
sceneView[row][2] * transform[2][col];
};
GXSetProjection(projection, GX_PERSPECTIVE);
GXSetCurrentMtx(GX_PNMTX0);
GXSetViewport(0, 0, 160, 120, 0, 1);
@@ -195,12 +216,28 @@ int main(int argc, char** argv) {
// Keep palette triangles small to limit fill cost during uniform stress.
const float extent = indexed ? 0.02f : 1.0f;
if (indexed) {
Mtx viewed;
applyView(viewed);
for (unsigned matrix = 0; matrix < 10; ++matrix)
GXLoadPosMtxImm(transform, matrix * 3);
GXLoadPosMtxImm(viewed, matrix * 3);
}
for (unsigned draw = 0; draw < drawCount; ++draw) {
transform[1][3] = static_cast<float>(draw % 20) * 0.01f;
GXLoadPosMtxImm(transform, GX_PNMTX0);
Mtx viewed;
applyView(viewed);
if (particles) {
// CPU-authored billboard centers, identity XF, current vertex shapes.
// Exercise the actual FIFO decoder, seal and late eye-uniform path.
Mtx identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
if (draw == 0) GXLoadPosMtxImm(identity, GX_PNMTX0);
GXBegin(GX_QUADS, GX_VTXFMT0, 4);
for (const auto& corner : {std::pair{-0.01f, -0.01f}, std::pair{0.01f, -0.01f},
std::pair{0.01f, 0.01f}, std::pair{-0.01f, 0.01f}})
GXPosition3f32(viewed[0][3] + corner.first, viewed[1][3] + corner.second, viewed[2][3]);
GXEnd();
continue;
}
GXLoadPosMtxImm(viewed, GX_PNMTX0);
GXBegin(GX_TRIANGLES, GX_VTXFMT0, indexed ? 30 : 3);
for (unsigned matrix = 0; matrix < (indexed ? 10u : 1u); ++matrix) {
if (indexed)
@@ -248,6 +285,6 @@ int main(int argc, char** argv) {
measuredSubmissions, elapsed, fps);
aurora_shutdown();
// More headset submissions must not come at the expense of simulation speed.
const double target = mode == 2 ? 75 : mode == 1 ? 90 : 60;
const double target = mode == 2 ? (headsetHz + 60) / 2.0 : mode == 1 ? headsetHz : 60;
return 240 / elapsed > 55 && fps > target - 5 && fps < target + 5 ? 0 : 1;
}
@@ -1,6 +1,60 @@
#include "stereo_interpolation.hpp"
#include "scene_camera.hpp"
#include <gtest/gtest.h>
#include <cmath>
#include <limits>
TEST(StereoInterpolation, SeparatesCameraFromHeldGeometryAndFirstPersonAnchor) {
using aurora::Mat3x4;
using aurora::gfx::stereo_replay::compose_affine;
const Mat3x4<float> identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
const float angle = 0.08f;
const float c = std::cos(angle), s = std::sin(angle);
const Mat3x4<float> currentView{{c, 0, s, 0}, {0, 1, 0, 0}, {-s, 0, c, 0}};
const Mat3x4<float> object{{1, 0, 0, 1000}, {0, 1, 0, 0}, {0, 0, 1, -100000}};
const auto currentObject = compose_affine(currentView, object);
aurora::stereo::SceneCameraMotion motion;
for (bool firstPerson : {false, true}) {
// A moving seat is distinct from the game's chase camera. Both must be
// sampled once, and only once, even for geometry with no usable history.
auto previousAnchor = identity, currentAnchor = identity;
if (firstPerson) {
previousAnchor.m1[3] = 200;
currentAnchor.m1[3] = 230;
}
ASSERT_TRUE(motion.prepare(identity, currentView, previousAnchor, currentAnchor));
for (float weight : {0.f, 1.f / 3, 2.f / 3, 1.f}) {
Mat3x4<float> sampledAnchor{}, expectedPose{}, expectedView{};
ASSERT_TRUE(motion.sample(weight, sampledAnchor));
ASSERT_TRUE(aurora::gx::interpolate_transform(motion.previousPose, motion.currentPose, weight, expectedPose));
ASSERT_TRUE(aurora::stereo::inverse_rigid_view(expectedPose, expectedView));
const auto expected = compose_affine(expectedView, object);
// Identical math covers held particle vertices baked into current view
// space and an unmatched/rejected billboard using its current matrix.
const auto actual = compose_affine(sampledAnchor, currentObject);
EXPECT_NEAR(actual.m0.w(), expected.m0.w(), 0.02f);
EXPECT_NEAR(actual.m1.w(), expected.m1.w(), 0.02f);
EXPECT_NEAR(actual.m2.w(), expected.m2.w(), 0.02f);
}
}
}
TEST(StereoInterpolation, CameraCutsAndMalformedViewsDisableCameraSeparation) {
const aurora::Mat3x4<float> identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}};
aurora::stereo::SceneCameraMotion motion;
ASSERT_TRUE(motion.prepare(identity, identity, identity, identity));
auto cut = identity;
cut.m0[3] = 2000;
EXPECT_FALSE(motion.prepare(identity, cut, identity, identity));
EXPECT_FALSE(motion.active);
cut = identity;
cut.m0[0] = 2;
EXPECT_FALSE(motion.prepare(identity, cut, identity, identity));
cut.m0[0] = std::numeric_limits<float>::quiet_NaN();
EXPECT_FALSE(motion.prepare(identity, cut, identity, identity));
cut = {{-1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, -1, 0}};
EXPECT_FALSE(motion.prepare(identity, cut, identity, identity));
}
TEST(StereoInterpolation, ContinuousMotionAcross60HzScenesAtHeadsetRates) {
constexpr uint64_t interval = 16'666'667;
@@ -30,3 +84,100 @@ TEST(StereoInterpolation, MissingTimingAndStallsDoNotExtrapolate) {
EXPECT_FLOAT_EQ(interpolation_weight(105, 100, 10), 0.5);
EXPECT_FLOAT_EQ(interpolation_weight(500, 100, 10), 1);
}
TEST(StereoInterpolation, PlaybackCadenceDoesNotFollowFutureHeadPrediction) {
constexpr uint64_t start = 1'000'000'000, interval = 16'666'667;
for (uint64_t hz : {72u, 90u, 120u}) {
// The producer's desktop presentation boundary can be ahead of, or behind,
// its actual seal. Neither offset belongs in headset scene playback.
for (int64_t scheduleOffset : {-5'000'000, 0, 7'000'000}) {
const uint64_t origin = start + scheduleOffset;
aurora::stereo::ScenePlaybackClock clock;
clock.begin_scene(origin, start, false);
uint64_t sealedScene = 0;
double previousPosition = 0;
uint32_t oldClampedSamples = 0;
for (uint64_t sample = 1; sample <= hz; ++sample) {
const uint64_t now = start + sample * 1'000'000'000 / hz;
const uint64_t scene = (now - start) / interval;
const uint64_t boundary = origin + scene * interval;
if (scene != sealedScene) {
// Seal jitter must not re-phase the entire playback clock.
clock.begin_scene(boundary, start + scene * interval + (scene % 3) * 100'000, true);
sealedScene = scene;
}
// The captured Virtual Desktop session predicted 37-65 ms ahead.
const uint64_t displayTime = now + (37 + sample % 29) * 1'000'000;
oldClampedSamples += aurora::stereo::interpolation_weight(displayTime, boundary, interval) == 1.0f;
const float weight = aurora::stereo::interpolation_weight(clock.sample_time(now), boundary, interval);
const double position = static_cast<double>(scene) + weight;
if (sample > 1)
EXPECT_NEAR(position - previousPosition, 1'000'000'000.0 / hz / interval, 1e-5);
previousPosition = position;
}
EXPECT_EQ(oldClampedSamples, hz); // Regression reproduces the old all-current result.
}
}
}
TEST(StereoInterpolation, PlaybackReanchorsOnCutsButNeverExtrapolatesAStalledScene) {
aurora::stereo::ScenePlaybackClock clock;
EXPECT_EQ(clock.sample_time(100), 0u);
clock.begin_scene(1000, 100, false);
EXPECT_EQ(clock.sample_time(105), 1005u);
clock.begin_scene(1010, 113, true);
EXPECT_EQ(clock.sample_time(115), 1015u); // Seal latency is not a new clock origin.
EXPECT_FLOAT_EQ(aurora::stereo::interpolation_weight(clock.sample_time(500), 1010, 10), 1);
clock.begin_scene(2000, 500, false); // Stall recovery / scene change.
EXPECT_EQ(clock.sample_time(505), 2005u);
clock.begin_scene(100, 510, true); // Reset producer schedule.
EXPECT_EQ(clock.sample_time(515), 105u);
clock.begin_scene(110, 20, true); // Reset host clock.
EXPECT_EQ(clock.sample_time(25), 115u);
EXPECT_EQ(clock.sample_time(19), 0u);
}
TEST(StereoInterpolation, EarlierProductionAfterWarmupDoesNotClampToPreviousEndpoints) {
aurora::stereo::ScenePlaybackClock clock;
clock.begin_scene(1000, 100, false);
clock.begin_scene(1010, 106, true); // Producer sheds four time units of warm-up latency.
EXPECT_EQ(clock.sample_time(106), 1010u);
EXPECT_EQ(clock.sample_time(111), 1015u);
clock.begin_scene(1020, 118, true); // A subsequent late seal must not move the clock back.
EXPECT_EQ(clock.sample_time(118), 1022u);
EXPECT_FLOAT_EQ(aurora::stereo::interpolation_weight(clock.sample_time(118), 1020, 10), 0.2f);
}
TEST(StereoInterpolation, MotionDiagnosticsDistinguishCadenceFromSubmissionCount) {
aurora::stereo::MotionSamples samples;
constexpr uint64_t boundary = 1'000'000'000, interval = 16'666'667;
const auto record = [&](uint64_t display, uint64_t sceneBoundary, bool continuous = true) {
samples.record(display, sceneBoundary, interval, continuous,
aurora::stereo::interpolation_weight(display, sceneBoundary, interval));
};
record(boundary, boundary);
record(boundary + interval / 2, boundary);
record(boundary + interval, boundary);
// A future display deadline outruns the retained scene; another submission
// cannot advance its motion, even though it can apply a fresh head pose.
record(boundary + 2 * interval, boundary);
record(boundary + 2 * interval, boundary + interval);
EXPECT_EQ(samples.samples, 5u);
EXPECT_EQ(samples.blended, 1u);
EXPECT_EQ(samples.atPrevious, 1u);
EXPECT_EQ(samples.atCurrent, 3u);
EXPECT_EQ(samples.repeated, 1u);
EXPECT_EQ(samples.backwards, 0u);
EXPECT_EQ(samples.minStep, 0u);
EXPECT_EQ(samples.maxStep, interval);
samples.clear_window();
record(boundary + 2 * interval, boundary + interval);
EXPECT_EQ(samples.samples, 1u);
EXPECT_EQ(samples.repeated, 1u); // Preserve cadence across reporting windows.
record(boundary + interval / 2, boundary);
EXPECT_EQ(samples.backwards, 1u);
record(boundary, boundary, false);
record(boundary, boundary);
EXPECT_EQ(samples.discontinuous, 1u);
EXPECT_EQ(samples.backwards, 1u); // A camera cut starts a new sequence.
}