From 9e37bb644789ca8bb6ff0e331f96dcd79398b335 Mon Sep 17 00:00:00 2001 From: iChris4 Date: Fri, 11 Sep 2026 02:46:23 +0200 Subject: [PATCH] Fixed VR Frame Interpolation frame-history copying bottleneck --- OPENXR.md | 11 ++ .../tests/stereo_frame_worker_smoke.cpp | 100 +++++++++++++++--- 2 files changed, 94 insertions(+), 17 deletions(-) diff --git a/OPENXR.md b/OPENXR.md index 0f17fe1..b5d5f74 100644 --- a/OPENXR.md +++ b/OPENXR.md @@ -252,6 +252,12 @@ produced 359 new stereo submissions in 4 seconds (89.7 FPS). This verifies submi not full-race performance or visual quality on a headset. Pass a draw count, for example `stereo_frame_worker_smoke 2000`, to stress uniform preparation and renderer/producer overlap; `stereo_frame_worker_smoke 2000 0` checks native stereo with interpolation Off. +Use `stereo_frame_worker_smoke 1000 1 1` to exercise ten-matrix palettes and their +larger uniform history, or `stereo_frame_worker_smoke 1000 2 1` to switch interpolation +On/Off during recording. The test compositor discards obsolete ticks and uses +high-resolution waits on Windows, keeping missed ticks from accumulating into bursts. +It pre-warms the next game frame like the runtime and excludes the first 60 frames +from timing so shader compilation and initial resource allocation do not skew steady-state results. The test checks that the producer stays above 55 FPS as well as checking headset submissions; replaying an old scene more often must not hide a slowed simulation. Validate actual races in VDXR at 90 Hz with Auto/90 selected, including race entry/exit, first person, recentering and pauses. @@ -263,6 +269,11 @@ Retained interpolation reserves eye ranges at seal time and fills them once at t time. VR interpolation also releases the producer after sealing so eye encoding can overlap the next game frame, as it does with desktop interpolation. +When VR interpolation is enabled at batch start, uniform recording also uses cached CPU +memory. Matching and history capture read that buffer, then the used prefix is copied to +the mapped upload buffer before unmapping. The backing choice stays fixed until the batch +ends, including mid-frame flushes, so live setting changes cannot invalidate pending tasks. + ## Current limitations - Only the project's supported PAL `RMCP01` translation has race instrumentation addresses. diff --git a/aurora-main/tests/stereo_frame_worker_smoke.cpp b/aurora-main/tests/stereo_frame_worker_smoke.cpp index 36204ec..a2d5faf 100644 --- a/aurora-main/tests/stereo_frame_worker_smoke.cpp +++ b/aurora-main/tests/stereo_frame_worker_smoke.cpp @@ -15,8 +15,39 @@ #include #include #include +#if defined(_WIN32) +#ifndef NOMINMAX +#define NOMINMAX +#endif +#include +#endif using Clock = std::chrono::steady_clock; +static void WaitUntil(Clock::time_point deadline) { +#if defined(_WIN32) + // Match the runtime's high-resolution pacing. Sleep's coarse Windows timer + // would otherwise turn an offscreen 90 Hz test into a roughly 60 Hz test. + struct Timer { + HANDLE handle = CreateWaitableTimerExW(nullptr, nullptr, CREATE_WAITABLE_TIMER_HIGH_RESOLUTION, + TIMER_MODIFY_STATE | SYNCHRONIZE); + ~Timer() { + if (handle != nullptr) + CloseHandle(handle); + } + }; + static thread_local Timer timer; + const auto remaining = std::chrono::duration_cast(deadline - Clock::now()).count(); + if (remaining <= 0) + return; + LARGE_INTEGER due{}; + due.QuadPart = -std::max(remaining / 100, 1); + if (timer.handle != nullptr && SetWaitableTimerEx(timer.handle, &due, 0, nullptr, nullptr, nullptr, 0)) { + WaitForSingleObject(timer.handle, INFINITE); + return; + } +#endif + std::this_thread::sleep_until(deadline); +} static std::mutex packetMutex; static std::condition_variable packetCv; static AuroraStereoFrame packet{}; @@ -24,6 +55,7 @@ static bool available = false; static bool stop = false; static uint64_t completed = 0; static std::atomic_uint32_t submitted{0}; +static uint64_t wakeLateness = 0, submitTime = 0, skippedTicks = 0, compositorFrames = 0; static bool Provide(uint32_t, AuroraStereoFrame* output, void*) { std::lock_guard lock(packetMutex); @@ -76,7 +108,10 @@ int main(int argc, char** argv) { uint64_t slot = 1; for (uint64_t token = 1;; ++token) { const auto deadline = start + std::chrono::nanoseconds(slot * 1'000'000'000 / 90); - std::this_thread::sleep_until(deadline); + WaitUntil(deadline); + const auto woke = Clock::now(); + wakeLateness += + std::max(0, std::chrono::duration_cast(woke - deadline).count()); { std::lock_guard lock(packetMutex); if (stop) @@ -84,8 +119,11 @@ int main(int argc, char** argv) { packet = {}; packet.frameToken = token; packet.contentTag = 42; + // Wake to render the next display tick, as xrWaitFrame does, rather + // than announcing an image whose display deadline has already passed. packet.displayTimeNanos = - std::chrono::duration_cast(deadline.time_since_epoch()).count(); + std::chrono::duration_cast(deadline.time_since_epoch()).count() + + 1'000'000'000 / 90; for (auto& eye : packet.eyes) { eye.width = 160; eye.height = 120; @@ -102,17 +140,33 @@ int main(int argc, char** argv) { packetCv.wait(lock, [&] { return stop || completed == token; }); if (stop) return; - // A real compositor advances to a future display time after a missed - // tick. Do not burst old deadlines when interpolation is re-enabled. + // Retain the current render tick (its display deadline is still ahead), + // but discard older ticks. Skipping even the current tick would insert + // an idle period whenever rendering overruns a wakeup by a fraction. const auto elapsed = std::chrono::duration_cast(Clock::now() - start).count(); - slot = std::max(slot + 1, static_cast(elapsed) * 90 / 1'000'000'000 + 1); + submitTime += std::chrono::duration_cast(Clock::now() - woke).count(); + ++compositorFrames; + const auto nextSlot = std::max(slot + 1, static_cast(elapsed) * 90 / 1'000'000'000); + skippedTicks += nextSlot - slot - 1; + slot = nextSlot; } }); - const auto start = Clock::now(); - for (uint64_t frame = 0; frame < 240; ++frame) { + constexpr uint64_t warmupFrames = 60; + auto start = Clock::now(); + auto measuredStart = start; + uint32_t initialSubmissions = 0; + for (uint64_t frame = 0; frame < warmupFrames + 240; ++frame) { + if (frame == warmupFrames) { + // Measure a running scene after shader compilation and initial resource + // creation, and rebase the producer so it owes no catch-up frames. + aurora_wait_for_frame_worker(); + measuredStart = Clock::now(); + start = measuredStart - std::chrono::nanoseconds(frame * 1'000'000'000 / 60); + initialSubmissions = submitted.load(); + } const auto boundary = start + std::chrono::nanoseconds((frame + 1) * 1'000'000'000 / 60); - std::this_thread::sleep_until(boundary); + WaitUntil(boundary); aurora_update(); if (!aurora_begin_frame()) continue; @@ -138,6 +192,8 @@ int main(int argc, char** argv) { GXSetNumTevStages(1); GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD_NULL, GX_TEXMAP_NULL, GX_COLOR_NULL); GXSetTevOp(GX_TEVSTAGE0, GX_PASSCLR); + // Keep palette triangles small to limit fill cost during uniform stress. + const float extent = indexed ? 0.02f : 1.0f; if (indexed) { for (unsigned matrix = 0; matrix < 10; ++matrix) GXLoadPosMtxImm(transform, matrix * 3); @@ -149,20 +205,25 @@ int main(int argc, char** argv) { for (unsigned matrix = 0; matrix < (indexed ? 10u : 1u); ++matrix) { if (indexed) GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3); - GXPosition3f32(-1 + static_cast(draw) * 0.0001f, -1, 0); + GXPosition3f32((-1 + static_cast(draw) * 0.0001f) * extent, -extent, 0); if (indexed) GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3); - GXPosition3f32(1, -1, 0); + GXPosition3f32(extent, -extent, 0); if (indexed) GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3); - GXPosition3f32(0, 1, 0); + GXPosition3f32(0, extent, 0); } GXEnd(); } aurora_end_frame_tagged(42); - if (drawCount > 1 && (frame + 1) % 60 == 0) { - std::printf("Completed %llu producer frames in %.2f s\n", static_cast(frame + 1), - std::chrono::duration(Clock::now() - start).count()); + // The runtime pre-warms immediately after its asynchronous seal. The + // worker needs this permit before publishing SEALED or encoding XR eyes. + // Waiting until the next 60 Hz tick would strand it for a whole interval. + aurora_begin_frame(); + if (drawCount > 1 && frame >= warmupFrames && (frame + 1) % 60 == 0) { + std::printf("Completed %llu producer frames in %.2f s\n", + static_cast(frame + 1 - warmupFrames), + std::chrono::duration(Clock::now() - measuredStart).count()); std::fflush(stdout); } } @@ -176,10 +237,15 @@ int main(int argc, char** argv) { aurora_quiesce_frame_worker(); aurora_set_stereo_frame_provider(nullptr, nullptr); aurora::stereo::set_sink(nullptr, nullptr); - const double elapsed = std::chrono::duration(Clock::now() - start).count(); - const double fps = submitted.load() / elapsed; + const double elapsed = std::chrono::duration(Clock::now() - measuredStart).count(); + const auto measuredSubmissions = submitted.load() - initialSubmissions; + const double fps = measuredSubmissions / elapsed; + if (compositorFrames != 0) + std::printf("Compositor (including warm-up): wake late %.2f ms; submit %.2f ms; skipped %llu ticks\n", + wakeLateness / (1.0e6 * compositorFrames), submitTime / (1.0e6 * compositorFrames), + static_cast(skippedTicks)); std::printf("%u draws: producer %.1f FPS; %u stereo submissions in %.2f s (%.1f FPS)\n", drawCount, 240 / elapsed, - submitted.load(), elapsed, fps); + measuredSubmissions, elapsed, fps); aurora_shutdown(); // More headset submissions must not come at the expense of simulation speed. const double target = mode == 2 ? 75 : mode == 1 ? 90 : 60;