Fixed VR Frame Interpolation frame-history copying bottleneck

This commit is contained in:
iChris4 committed 2026-09-11 02:46:23 +02:00
1 parent 023c7dd082
commit 9e37bb6447
2 files changed
+94 -17

No files matched your search

+83 -17
View File
@@ -15,8 +15,39 @@
#include <filesystem>
#include <mutex>
#include <thread>
#if defined(_WIN32)
#ifndef NOMINMAX
#define NOMINMAX
#endif
#include <windows.h>
#endif
using Clock = std::chrono::steady_clock;
static void WaitUntil(Clock::time_point deadline) {
#if defined(_WIN32)
// Match the runtime's high-resolution pacing. Sleep's coarse Windows timer
// would otherwise turn an offscreen 90 Hz test into a roughly 60 Hz test.
struct Timer {
HANDLE handle = CreateWaitableTimerExW(nullptr, nullptr, CREATE_WAITABLE_TIMER_HIGH_RESOLUTION,
TIMER_MODIFY_STATE | SYNCHRONIZE);
~Timer() {
if (handle != nullptr)
CloseHandle(handle);
}
};
static thread_local Timer timer;
const auto remaining = std::chrono::duration_cast<std::chrono::nanoseconds>(deadline - Clock::now()).count();
if (remaining <= 0)
return;
LARGE_INTEGER due{};
due.QuadPart = -std::max<int64_t>(remaining / 100, 1);
if (timer.handle != nullptr && SetWaitableTimerEx(timer.handle, &due, 0, nullptr, nullptr, nullptr, 0)) {
WaitForSingleObject(timer.handle, INFINITE);
return;
}
#endif
std::this_thread::sleep_until(deadline);
}
static std::mutex packetMutex;
static std::condition_variable packetCv;
static AuroraStereoFrame packet{};
@@ -24,6 +55,7 @@ static bool available = false;
static bool stop = false;
static uint64_t completed = 0;
static std::atomic_uint32_t submitted{0};
static uint64_t wakeLateness = 0, submitTime = 0, skippedTicks = 0, compositorFrames = 0;
static bool Provide(uint32_t, AuroraStereoFrame* output, void*) {
std::lock_guard lock(packetMutex);
@@ -76,7 +108,10 @@ int main(int argc, char** argv) {
uint64_t slot = 1;
for (uint64_t token = 1;; ++token) {
const auto deadline = start + std::chrono::nanoseconds(slot * 1'000'000'000 / 90);
std::this_thread::sleep_until(deadline);
WaitUntil(deadline);
const auto woke = Clock::now();
wakeLateness +=
std::max<int64_t>(0, std::chrono::duration_cast<std::chrono::nanoseconds>(woke - deadline).count());
{
std::lock_guard lock(packetMutex);
if (stop)
@@ -84,8 +119,11 @@ int main(int argc, char** argv) {
packet = {};
packet.frameToken = token;
packet.contentTag = 42;
// Wake to render the next display tick, as xrWaitFrame does, rather
// than announcing an image whose display deadline has already passed.
packet.displayTimeNanos =
std::chrono::duration_cast<std::chrono::nanoseconds>(deadline.time_since_epoch()).count();
std::chrono::duration_cast<std::chrono::nanoseconds>(deadline.time_since_epoch()).count() +
1'000'000'000 / 90;
for (auto& eye : packet.eyes) {
eye.width = 160;
eye.height = 120;
@@ -102,17 +140,33 @@ int main(int argc, char** argv) {
packetCv.wait(lock, [&] { return stop || completed == token; });
if (stop)
return;
// A real compositor advances to a future display time after a missed
// tick. Do not burst old deadlines when interpolation is re-enabled.
// Retain the current render tick (its display deadline is still ahead),
// but discard older ticks. Skipping even the current tick would insert
// an idle period whenever rendering overruns a wakeup by a fraction.
const auto elapsed = std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - start).count();
slot = std::max(slot + 1, static_cast<uint64_t>(elapsed) * 90 / 1'000'000'000 + 1);
submitTime += std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - woke).count();
++compositorFrames;
const auto nextSlot = std::max(slot + 1, static_cast<uint64_t>(elapsed) * 90 / 1'000'000'000);
skippedTicks += nextSlot - slot - 1;
slot = nextSlot;
}
});
const auto start = Clock::now();
for (uint64_t frame = 0; frame < 240; ++frame) {
constexpr uint64_t warmupFrames = 60;
auto start = Clock::now();
auto measuredStart = start;
uint32_t initialSubmissions = 0;
for (uint64_t frame = 0; frame < warmupFrames + 240; ++frame) {
if (frame == warmupFrames) {
// Measure a running scene after shader compilation and initial resource
// creation, and rebase the producer so it owes no catch-up frames.
aurora_wait_for_frame_worker();
measuredStart = Clock::now();
start = measuredStart - std::chrono::nanoseconds(frame * 1'000'000'000 / 60);
initialSubmissions = submitted.load();
}
const auto boundary = start + std::chrono::nanoseconds((frame + 1) * 1'000'000'000 / 60);
std::this_thread::sleep_until(boundary);
WaitUntil(boundary);
aurora_update();
if (!aurora_begin_frame())
continue;
@@ -138,6 +192,8 @@ int main(int argc, char** argv) {
GXSetNumTevStages(1);
GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD_NULL, GX_TEXMAP_NULL, GX_COLOR_NULL);
GXSetTevOp(GX_TEVSTAGE0, GX_PASSCLR);
// Keep palette triangles small to limit fill cost during uniform stress.
const float extent = indexed ? 0.02f : 1.0f;
if (indexed) {
for (unsigned matrix = 0; matrix < 10; ++matrix)
GXLoadPosMtxImm(transform, matrix * 3);
@@ -149,20 +205,25 @@ int main(int argc, char** argv) {
for (unsigned matrix = 0; matrix < (indexed ? 10u : 1u); ++matrix) {
if (indexed)
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
GXPosition3f32(-1 + static_cast<float>(draw) * 0.0001f, -1, 0);
GXPosition3f32((-1 + static_cast<float>(draw) * 0.0001f) * extent, -extent, 0);
if (indexed)
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
GXPosition3f32(1, -1, 0);
GXPosition3f32(extent, -extent, 0);
if (indexed)
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
GXPosition3f32(0, 1, 0);
GXPosition3f32(0, extent, 0);
}
GXEnd();
}
aurora_end_frame_tagged(42);
if (drawCount > 1 && (frame + 1) % 60 == 0) {
std::printf("Completed %llu producer frames in %.2f s\n", static_cast<unsigned long long>(frame + 1),
std::chrono::duration<double>(Clock::now() - start).count());
// The runtime pre-warms immediately after its asynchronous seal. The
// worker needs this permit before publishing SEALED or encoding XR eyes.
// Waiting until the next 60 Hz tick would strand it for a whole interval.
aurora_begin_frame();
if (drawCount > 1 && frame >= warmupFrames && (frame + 1) % 60 == 0) {
std::printf("Completed %llu producer frames in %.2f s\n",
static_cast<unsigned long long>(frame + 1 - warmupFrames),
std::chrono::duration<double>(Clock::now() - measuredStart).count());
std::fflush(stdout);
}
}
@@ -176,10 +237,15 @@ int main(int argc, char** argv) {
aurora_quiesce_frame_worker();
aurora_set_stereo_frame_provider(nullptr, nullptr);
aurora::stereo::set_sink(nullptr, nullptr);
const double elapsed = std::chrono::duration<double>(Clock::now() - start).count();
const double fps = submitted.load() / elapsed;
const double elapsed = std::chrono::duration<double>(Clock::now() - measuredStart).count();
const auto measuredSubmissions = submitted.load() - initialSubmissions;
const double fps = measuredSubmissions / elapsed;
if (compositorFrames != 0)
std::printf("Compositor (including warm-up): wake late %.2f ms; submit %.2f ms; skipped %llu ticks\n",
wakeLateness / (1.0e6 * compositorFrames), submitTime / (1.0e6 * compositorFrames),
static_cast<unsigned long long>(skippedTicks));
std::printf("%u draws: producer %.1f FPS; %u stereo submissions in %.2f s (%.1f FPS)\n", drawCount, 240 / elapsed,
submitted.load(), elapsed, fps);
measuredSubmissions, elapsed, fps);
aurora_shutdown();
// More headset submissions must not come at the expense of simulation speed.
const double target = mode == 2 ? 75 : mode == 1 ? 90 : 60;