Fixed VR Frame Interpolation frame-history copying bottleneck

This commit is contained in:
iChris4 committed 2026-09-11 02:46:23 +02:00
1 parent 023c7dd082
commit 9e37bb6447
2 files changed
+94 -17

No files matched your search

+11
View File
@@ -252,6 +252,12 @@ produced 359 new stereo submissions in 4 seconds (89.7 FPS). This verifies submi
not full-race performance or visual quality on a headset. Pass a draw count, for example not full-race performance or visual quality on a headset. Pass a draw count, for example
`stereo_frame_worker_smoke 2000`, to stress uniform preparation and renderer/producer overlap; `stereo_frame_worker_smoke 2000`, to stress uniform preparation and renderer/producer overlap;
`stereo_frame_worker_smoke 2000 0` checks native stereo with interpolation Off. `stereo_frame_worker_smoke 2000 0` checks native stereo with interpolation Off.
Use `stereo_frame_worker_smoke 1000 1 1` to exercise ten-matrix palettes and their
larger uniform history, or `stereo_frame_worker_smoke 1000 2 1` to switch interpolation
On/Off during recording. The test compositor discards obsolete ticks and uses
high-resolution waits on Windows, keeping missed ticks from accumulating into bursts.
It pre-warms the next game frame like the runtime and excludes the first 60 frames
from timing so shader compilation and initial resource allocation do not skew steady-state results.
The test checks that the producer stays above 55 FPS as well as checking headset submissions; The test checks that the producer stays above 55 FPS as well as checking headset submissions;
replaying an old scene more often must not hide a slowed simulation. Validate actual races in VDXR at 90 Hz with replaying an old scene more often must not hide a slowed simulation. Validate actual races in VDXR at 90 Hz with
Auto/90 selected, including race entry/exit, first person, recentering and pauses. Auto/90 selected, including race entry/exit, first person, recentering and pauses.
@@ -263,6 +269,11 @@ Retained interpolation reserves eye ranges at seal time and fills them once at t
time. VR interpolation also releases the producer after sealing so eye encoding can overlap the time. VR interpolation also releases the producer after sealing so eye encoding can overlap the
next game frame, as it does with desktop interpolation. next game frame, as it does with desktop interpolation.
When VR interpolation is enabled at batch start, uniform recording also uses cached CPU
memory. Matching and history capture read that buffer, then the used prefix is copied to
the mapped upload buffer before unmapping. The backing choice stays fixed until the batch
ends, including mid-frame flushes, so live setting changes cannot invalidate pending tasks.
## Current limitations ## Current limitations
- Only the project's supported PAL `RMCP01` translation has race instrumentation addresses. - Only the project's supported PAL `RMCP01` translation has race instrumentation addresses.
+83 -17
View File
@@ -15,8 +15,39 @@
#include <filesystem> #include <filesystem>
#include <mutex> #include <mutex>
#include <thread> #include <thread>
#if defined(_WIN32)
#ifndef NOMINMAX
#define NOMINMAX
#endif
#include <windows.h>
#endif
using Clock = std::chrono::steady_clock; using Clock = std::chrono::steady_clock;
static void WaitUntil(Clock::time_point deadline) {
#if defined(_WIN32)
// Match the runtime's high-resolution pacing. Sleep's coarse Windows timer
// would otherwise turn an offscreen 90 Hz test into a roughly 60 Hz test.
struct Timer {
HANDLE handle = CreateWaitableTimerExW(nullptr, nullptr, CREATE_WAITABLE_TIMER_HIGH_RESOLUTION,
TIMER_MODIFY_STATE | SYNCHRONIZE);
~Timer() {
if (handle != nullptr)
CloseHandle(handle);
}
};
static thread_local Timer timer;
const auto remaining = std::chrono::duration_cast<std::chrono::nanoseconds>(deadline - Clock::now()).count();
if (remaining <= 0)
return;
LARGE_INTEGER due{};
due.QuadPart = -std::max<int64_t>(remaining / 100, 1);
if (timer.handle != nullptr && SetWaitableTimerEx(timer.handle, &due, 0, nullptr, nullptr, nullptr, 0)) {
WaitForSingleObject(timer.handle, INFINITE);
return;
}
#endif
std::this_thread::sleep_until(deadline);
}
static std::mutex packetMutex; static std::mutex packetMutex;
static std::condition_variable packetCv; static std::condition_variable packetCv;
static AuroraStereoFrame packet{}; static AuroraStereoFrame packet{};
@@ -24,6 +55,7 @@ static bool available = false;
static bool stop = false; static bool stop = false;
static uint64_t completed = 0; static uint64_t completed = 0;
static std::atomic_uint32_t submitted{0}; static std::atomic_uint32_t submitted{0};
static uint64_t wakeLateness = 0, submitTime = 0, skippedTicks = 0, compositorFrames = 0;
static bool Provide(uint32_t, AuroraStereoFrame* output, void*) { static bool Provide(uint32_t, AuroraStereoFrame* output, void*) {
std::lock_guard lock(packetMutex); std::lock_guard lock(packetMutex);
@@ -76,7 +108,10 @@ int main(int argc, char** argv) {
uint64_t slot = 1; uint64_t slot = 1;
for (uint64_t token = 1;; ++token) { for (uint64_t token = 1;; ++token) {
const auto deadline = start + std::chrono::nanoseconds(slot * 1'000'000'000 / 90); const auto deadline = start + std::chrono::nanoseconds(slot * 1'000'000'000 / 90);
std::this_thread::sleep_until(deadline); WaitUntil(deadline);
const auto woke = Clock::now();
wakeLateness +=
std::max<int64_t>(0, std::chrono::duration_cast<std::chrono::nanoseconds>(woke - deadline).count());
{ {
std::lock_guard lock(packetMutex); std::lock_guard lock(packetMutex);
if (stop) if (stop)
@@ -84,8 +119,11 @@ int main(int argc, char** argv) {
packet = {}; packet = {};
packet.frameToken = token; packet.frameToken = token;
packet.contentTag = 42; packet.contentTag = 42;
// Wake to render the next display tick, as xrWaitFrame does, rather
// than announcing an image whose display deadline has already passed.
packet.displayTimeNanos = packet.displayTimeNanos =
std::chrono::duration_cast<std::chrono::nanoseconds>(deadline.time_since_epoch()).count(); std::chrono::duration_cast<std::chrono::nanoseconds>(deadline.time_since_epoch()).count() +
1'000'000'000 / 90;
for (auto& eye : packet.eyes) { for (auto& eye : packet.eyes) {
eye.width = 160; eye.width = 160;
eye.height = 120; eye.height = 120;
@@ -102,17 +140,33 @@ int main(int argc, char** argv) {
packetCv.wait(lock, [&] { return stop || completed == token; }); packetCv.wait(lock, [&] { return stop || completed == token; });
if (stop) if (stop)
return; return;
// A real compositor advances to a future display time after a missed // Retain the current render tick (its display deadline is still ahead),
// tick. Do not burst old deadlines when interpolation is re-enabled. // but discard older ticks. Skipping even the current tick would insert
// an idle period whenever rendering overruns a wakeup by a fraction.
const auto elapsed = std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - start).count(); const auto elapsed = std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - start).count();
slot = std::max(slot + 1, static_cast<uint64_t>(elapsed) * 90 / 1'000'000'000 + 1); submitTime += std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - woke).count();
++compositorFrames;
const auto nextSlot = std::max(slot + 1, static_cast<uint64_t>(elapsed) * 90 / 1'000'000'000);
skippedTicks += nextSlot - slot - 1;
slot = nextSlot;
} }
}); });
const auto start = Clock::now(); constexpr uint64_t warmupFrames = 60;
for (uint64_t frame = 0; frame < 240; ++frame) { auto start = Clock::now();
auto measuredStart = start;
uint32_t initialSubmissions = 0;
for (uint64_t frame = 0; frame < warmupFrames + 240; ++frame) {
if (frame == warmupFrames) {
// Measure a running scene after shader compilation and initial resource
// creation, and rebase the producer so it owes no catch-up frames.
aurora_wait_for_frame_worker();
measuredStart = Clock::now();
start = measuredStart - std::chrono::nanoseconds(frame * 1'000'000'000 / 60);
initialSubmissions = submitted.load();
}
const auto boundary = start + std::chrono::nanoseconds((frame + 1) * 1'000'000'000 / 60); const auto boundary = start + std::chrono::nanoseconds((frame + 1) * 1'000'000'000 / 60);
std::this_thread::sleep_until(boundary); WaitUntil(boundary);
aurora_update(); aurora_update();
if (!aurora_begin_frame()) if (!aurora_begin_frame())
continue; continue;
@@ -138,6 +192,8 @@ int main(int argc, char** argv) {
GXSetNumTevStages(1); GXSetNumTevStages(1);
GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD_NULL, GX_TEXMAP_NULL, GX_COLOR_NULL); GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD_NULL, GX_TEXMAP_NULL, GX_COLOR_NULL);
GXSetTevOp(GX_TEVSTAGE0, GX_PASSCLR); GXSetTevOp(GX_TEVSTAGE0, GX_PASSCLR);
// Keep palette triangles small to limit fill cost during uniform stress.
const float extent = indexed ? 0.02f : 1.0f;
if (indexed) { if (indexed) {
for (unsigned matrix = 0; matrix < 10; ++matrix) for (unsigned matrix = 0; matrix < 10; ++matrix)
GXLoadPosMtxImm(transform, matrix * 3); GXLoadPosMtxImm(transform, matrix * 3);
@@ -149,20 +205,25 @@ int main(int argc, char** argv) {
for (unsigned matrix = 0; matrix < (indexed ? 10u : 1u); ++matrix) { for (unsigned matrix = 0; matrix < (indexed ? 10u : 1u); ++matrix) {
if (indexed) if (indexed)
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3); GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
GXPosition3f32(-1 + static_cast<float>(draw) * 0.0001f, -1, 0); GXPosition3f32((-1 + static_cast<float>(draw) * 0.0001f) * extent, -extent, 0);
if (indexed) if (indexed)
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3); GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
GXPosition3f32(1, -1, 0); GXPosition3f32(extent, -extent, 0);
if (indexed) if (indexed)
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3); GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
GXPosition3f32(0, 1, 0); GXPosition3f32(0, extent, 0);
} }
GXEnd(); GXEnd();
} }
aurora_end_frame_tagged(42); aurora_end_frame_tagged(42);
if (drawCount > 1 && (frame + 1) % 60 == 0) { // The runtime pre-warms immediately after its asynchronous seal. The
std::printf("Completed %llu producer frames in %.2f s\n", static_cast<unsigned long long>(frame + 1), // worker needs this permit before publishing SEALED or encoding XR eyes.
std::chrono::duration<double>(Clock::now() - start).count()); // Waiting until the next 60 Hz tick would strand it for a whole interval.
aurora_begin_frame();
if (drawCount > 1 && frame >= warmupFrames && (frame + 1) % 60 == 0) {
std::printf("Completed %llu producer frames in %.2f s\n",
static_cast<unsigned long long>(frame + 1 - warmupFrames),
std::chrono::duration<double>(Clock::now() - measuredStart).count());
std::fflush(stdout); std::fflush(stdout);
} }
} }
@@ -176,10 +237,15 @@ int main(int argc, char** argv) {
aurora_quiesce_frame_worker(); aurora_quiesce_frame_worker();
aurora_set_stereo_frame_provider(nullptr, nullptr); aurora_set_stereo_frame_provider(nullptr, nullptr);
aurora::stereo::set_sink(nullptr, nullptr); aurora::stereo::set_sink(nullptr, nullptr);
const double elapsed = std::chrono::duration<double>(Clock::now() - start).count(); const double elapsed = std::chrono::duration<double>(Clock::now() - measuredStart).count();
const double fps = submitted.load() / elapsed; const auto measuredSubmissions = submitted.load() - initialSubmissions;
const double fps = measuredSubmissions / elapsed;
if (compositorFrames != 0)
std::printf("Compositor (including warm-up): wake late %.2f ms; submit %.2f ms; skipped %llu ticks\n",
wakeLateness / (1.0e6 * compositorFrames), submitTime / (1.0e6 * compositorFrames),
static_cast<unsigned long long>(skippedTicks));
std::printf("%u draws: producer %.1f FPS; %u stereo submissions in %.2f s (%.1f FPS)\n", drawCount, 240 / elapsed, std::printf("%u draws: producer %.1f FPS; %u stereo submissions in %.2f s (%.1f FPS)\n", drawCount, 240 / elapsed,
submitted.load(), elapsed, fps); measuredSubmissions, elapsed, fps);
aurora_shutdown(); aurora_shutdown();
// More headset submissions must not come at the expense of simulation speed. // More headset submissions must not come at the expense of simulation speed.
const double target = mode == 2 ? 75 : mode == 1 ? 90 : 60; const double target = mode == 2 ? 75 : mode == 1 ? 90 : 60;