mirror of
https://github.com/mitch030504/Wiicompiled_VR_Frame.git
synced 2026-10-06 04:04:18 +02:00
Fixed VR Frame Interpolation frame-history copying bottleneck
This commit is contained in:
1 parent
023c7dd082
commit
9e37bb6447
2 files changed
+94
-17
No files matched your search
@@ -252,6 +252,12 @@ produced 359 new stereo submissions in 4 seconds (89.7 FPS). This verifies submi
|
|||||||
not full-race performance or visual quality on a headset. Pass a draw count, for example
|
not full-race performance or visual quality on a headset. Pass a draw count, for example
|
||||||
`stereo_frame_worker_smoke 2000`, to stress uniform preparation and renderer/producer overlap;
|
`stereo_frame_worker_smoke 2000`, to stress uniform preparation and renderer/producer overlap;
|
||||||
`stereo_frame_worker_smoke 2000 0` checks native stereo with interpolation Off.
|
`stereo_frame_worker_smoke 2000 0` checks native stereo with interpolation Off.
|
||||||
|
Use `stereo_frame_worker_smoke 1000 1 1` to exercise ten-matrix palettes and their
|
||||||
|
larger uniform history, or `stereo_frame_worker_smoke 1000 2 1` to switch interpolation
|
||||||
|
On/Off during recording. The test compositor discards obsolete ticks and uses
|
||||||
|
high-resolution waits on Windows, keeping missed ticks from accumulating into bursts.
|
||||||
|
It pre-warms the next game frame like the runtime and excludes the first 60 frames
|
||||||
|
from timing so shader compilation and initial resource allocation do not skew steady-state results.
|
||||||
The test checks that the producer stays above 55 FPS as well as checking headset submissions;
|
The test checks that the producer stays above 55 FPS as well as checking headset submissions;
|
||||||
replaying an old scene more often must not hide a slowed simulation. Validate actual races in VDXR at 90 Hz with
|
replaying an old scene more often must not hide a slowed simulation. Validate actual races in VDXR at 90 Hz with
|
||||||
Auto/90 selected, including race entry/exit, first person, recentering and pauses.
|
Auto/90 selected, including race entry/exit, first person, recentering and pauses.
|
||||||
@@ -263,6 +269,11 @@ Retained interpolation reserves eye ranges at seal time and fills them once at t
|
|||||||
time. VR interpolation also releases the producer after sealing so eye encoding can overlap the
|
time. VR interpolation also releases the producer after sealing so eye encoding can overlap the
|
||||||
next game frame, as it does with desktop interpolation.
|
next game frame, as it does with desktop interpolation.
|
||||||
|
|
||||||
|
When VR interpolation is enabled at batch start, uniform recording also uses cached CPU
|
||||||
|
memory. Matching and history capture read that buffer, then the used prefix is copied to
|
||||||
|
the mapped upload buffer before unmapping. The backing choice stays fixed until the batch
|
||||||
|
ends, including mid-frame flushes, so live setting changes cannot invalidate pending tasks.
|
||||||
|
|
||||||
## Current limitations
|
## Current limitations
|
||||||
|
|
||||||
- Only the project's supported PAL `RMCP01` translation has race instrumentation addresses.
|
- Only the project's supported PAL `RMCP01` translation has race instrumentation addresses.
|
||||||
|
|||||||
@@ -15,8 +15,39 @@
|
|||||||
#include <filesystem>
|
#include <filesystem>
|
||||||
#include <mutex>
|
#include <mutex>
|
||||||
#include <thread>
|
#include <thread>
|
||||||
|
#if defined(_WIN32)
|
||||||
|
#ifndef NOMINMAX
|
||||||
|
#define NOMINMAX
|
||||||
|
#endif
|
||||||
|
#include <windows.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
using Clock = std::chrono::steady_clock;
|
using Clock = std::chrono::steady_clock;
|
||||||
|
static void WaitUntil(Clock::time_point deadline) {
|
||||||
|
#if defined(_WIN32)
|
||||||
|
// Match the runtime's high-resolution pacing. Sleep's coarse Windows timer
|
||||||
|
// would otherwise turn an offscreen 90 Hz test into a roughly 60 Hz test.
|
||||||
|
struct Timer {
|
||||||
|
HANDLE handle = CreateWaitableTimerExW(nullptr, nullptr, CREATE_WAITABLE_TIMER_HIGH_RESOLUTION,
|
||||||
|
TIMER_MODIFY_STATE | SYNCHRONIZE);
|
||||||
|
~Timer() {
|
||||||
|
if (handle != nullptr)
|
||||||
|
CloseHandle(handle);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
static thread_local Timer timer;
|
||||||
|
const auto remaining = std::chrono::duration_cast<std::chrono::nanoseconds>(deadline - Clock::now()).count();
|
||||||
|
if (remaining <= 0)
|
||||||
|
return;
|
||||||
|
LARGE_INTEGER due{};
|
||||||
|
due.QuadPart = -std::max<int64_t>(remaining / 100, 1);
|
||||||
|
if (timer.handle != nullptr && SetWaitableTimerEx(timer.handle, &due, 0, nullptr, nullptr, nullptr, 0)) {
|
||||||
|
WaitForSingleObject(timer.handle, INFINITE);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
std::this_thread::sleep_until(deadline);
|
||||||
|
}
|
||||||
static std::mutex packetMutex;
|
static std::mutex packetMutex;
|
||||||
static std::condition_variable packetCv;
|
static std::condition_variable packetCv;
|
||||||
static AuroraStereoFrame packet{};
|
static AuroraStereoFrame packet{};
|
||||||
@@ -24,6 +55,7 @@ static bool available = false;
|
|||||||
static bool stop = false;
|
static bool stop = false;
|
||||||
static uint64_t completed = 0;
|
static uint64_t completed = 0;
|
||||||
static std::atomic_uint32_t submitted{0};
|
static std::atomic_uint32_t submitted{0};
|
||||||
|
static uint64_t wakeLateness = 0, submitTime = 0, skippedTicks = 0, compositorFrames = 0;
|
||||||
|
|
||||||
static bool Provide(uint32_t, AuroraStereoFrame* output, void*) {
|
static bool Provide(uint32_t, AuroraStereoFrame* output, void*) {
|
||||||
std::lock_guard lock(packetMutex);
|
std::lock_guard lock(packetMutex);
|
||||||
@@ -76,7 +108,10 @@ int main(int argc, char** argv) {
|
|||||||
uint64_t slot = 1;
|
uint64_t slot = 1;
|
||||||
for (uint64_t token = 1;; ++token) {
|
for (uint64_t token = 1;; ++token) {
|
||||||
const auto deadline = start + std::chrono::nanoseconds(slot * 1'000'000'000 / 90);
|
const auto deadline = start + std::chrono::nanoseconds(slot * 1'000'000'000 / 90);
|
||||||
std::this_thread::sleep_until(deadline);
|
WaitUntil(deadline);
|
||||||
|
const auto woke = Clock::now();
|
||||||
|
wakeLateness +=
|
||||||
|
std::max<int64_t>(0, std::chrono::duration_cast<std::chrono::nanoseconds>(woke - deadline).count());
|
||||||
{
|
{
|
||||||
std::lock_guard lock(packetMutex);
|
std::lock_guard lock(packetMutex);
|
||||||
if (stop)
|
if (stop)
|
||||||
@@ -84,8 +119,11 @@ int main(int argc, char** argv) {
|
|||||||
packet = {};
|
packet = {};
|
||||||
packet.frameToken = token;
|
packet.frameToken = token;
|
||||||
packet.contentTag = 42;
|
packet.contentTag = 42;
|
||||||
|
// Wake to render the next display tick, as xrWaitFrame does, rather
|
||||||
|
// than announcing an image whose display deadline has already passed.
|
||||||
packet.displayTimeNanos =
|
packet.displayTimeNanos =
|
||||||
std::chrono::duration_cast<std::chrono::nanoseconds>(deadline.time_since_epoch()).count();
|
std::chrono::duration_cast<std::chrono::nanoseconds>(deadline.time_since_epoch()).count() +
|
||||||
|
1'000'000'000 / 90;
|
||||||
for (auto& eye : packet.eyes) {
|
for (auto& eye : packet.eyes) {
|
||||||
eye.width = 160;
|
eye.width = 160;
|
||||||
eye.height = 120;
|
eye.height = 120;
|
||||||
@@ -102,17 +140,33 @@ int main(int argc, char** argv) {
|
|||||||
packetCv.wait(lock, [&] { return stop || completed == token; });
|
packetCv.wait(lock, [&] { return stop || completed == token; });
|
||||||
if (stop)
|
if (stop)
|
||||||
return;
|
return;
|
||||||
// A real compositor advances to a future display time after a missed
|
// Retain the current render tick (its display deadline is still ahead),
|
||||||
// tick. Do not burst old deadlines when interpolation is re-enabled.
|
// but discard older ticks. Skipping even the current tick would insert
|
||||||
|
// an idle period whenever rendering overruns a wakeup by a fraction.
|
||||||
const auto elapsed = std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - start).count();
|
const auto elapsed = std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - start).count();
|
||||||
slot = std::max(slot + 1, static_cast<uint64_t>(elapsed) * 90 / 1'000'000'000 + 1);
|
submitTime += std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - woke).count();
|
||||||
|
++compositorFrames;
|
||||||
|
const auto nextSlot = std::max(slot + 1, static_cast<uint64_t>(elapsed) * 90 / 1'000'000'000);
|
||||||
|
skippedTicks += nextSlot - slot - 1;
|
||||||
|
slot = nextSlot;
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|
||||||
const auto start = Clock::now();
|
constexpr uint64_t warmupFrames = 60;
|
||||||
for (uint64_t frame = 0; frame < 240; ++frame) {
|
auto start = Clock::now();
|
||||||
|
auto measuredStart = start;
|
||||||
|
uint32_t initialSubmissions = 0;
|
||||||
|
for (uint64_t frame = 0; frame < warmupFrames + 240; ++frame) {
|
||||||
|
if (frame == warmupFrames) {
|
||||||
|
// Measure a running scene after shader compilation and initial resource
|
||||||
|
// creation, and rebase the producer so it owes no catch-up frames.
|
||||||
|
aurora_wait_for_frame_worker();
|
||||||
|
measuredStart = Clock::now();
|
||||||
|
start = measuredStart - std::chrono::nanoseconds(frame * 1'000'000'000 / 60);
|
||||||
|
initialSubmissions = submitted.load();
|
||||||
|
}
|
||||||
const auto boundary = start + std::chrono::nanoseconds((frame + 1) * 1'000'000'000 / 60);
|
const auto boundary = start + std::chrono::nanoseconds((frame + 1) * 1'000'000'000 / 60);
|
||||||
std::this_thread::sleep_until(boundary);
|
WaitUntil(boundary);
|
||||||
aurora_update();
|
aurora_update();
|
||||||
if (!aurora_begin_frame())
|
if (!aurora_begin_frame())
|
||||||
continue;
|
continue;
|
||||||
@@ -138,6 +192,8 @@ int main(int argc, char** argv) {
|
|||||||
GXSetNumTevStages(1);
|
GXSetNumTevStages(1);
|
||||||
GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD_NULL, GX_TEXMAP_NULL, GX_COLOR_NULL);
|
GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD_NULL, GX_TEXMAP_NULL, GX_COLOR_NULL);
|
||||||
GXSetTevOp(GX_TEVSTAGE0, GX_PASSCLR);
|
GXSetTevOp(GX_TEVSTAGE0, GX_PASSCLR);
|
||||||
|
// Keep palette triangles small to limit fill cost during uniform stress.
|
||||||
|
const float extent = indexed ? 0.02f : 1.0f;
|
||||||
if (indexed) {
|
if (indexed) {
|
||||||
for (unsigned matrix = 0; matrix < 10; ++matrix)
|
for (unsigned matrix = 0; matrix < 10; ++matrix)
|
||||||
GXLoadPosMtxImm(transform, matrix * 3);
|
GXLoadPosMtxImm(transform, matrix * 3);
|
||||||
@@ -149,20 +205,25 @@ int main(int argc, char** argv) {
|
|||||||
for (unsigned matrix = 0; matrix < (indexed ? 10u : 1u); ++matrix) {
|
for (unsigned matrix = 0; matrix < (indexed ? 10u : 1u); ++matrix) {
|
||||||
if (indexed)
|
if (indexed)
|
||||||
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
|
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
|
||||||
GXPosition3f32(-1 + static_cast<float>(draw) * 0.0001f, -1, 0);
|
GXPosition3f32((-1 + static_cast<float>(draw) * 0.0001f) * extent, -extent, 0);
|
||||||
if (indexed)
|
if (indexed)
|
||||||
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
|
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
|
||||||
GXPosition3f32(1, -1, 0);
|
GXPosition3f32(extent, -extent, 0);
|
||||||
if (indexed)
|
if (indexed)
|
||||||
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
|
GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3);
|
||||||
GXPosition3f32(0, 1, 0);
|
GXPosition3f32(0, extent, 0);
|
||||||
}
|
}
|
||||||
GXEnd();
|
GXEnd();
|
||||||
}
|
}
|
||||||
aurora_end_frame_tagged(42);
|
aurora_end_frame_tagged(42);
|
||||||
if (drawCount > 1 && (frame + 1) % 60 == 0) {
|
// The runtime pre-warms immediately after its asynchronous seal. The
|
||||||
std::printf("Completed %llu producer frames in %.2f s\n", static_cast<unsigned long long>(frame + 1),
|
// worker needs this permit before publishing SEALED or encoding XR eyes.
|
||||||
std::chrono::duration<double>(Clock::now() - start).count());
|
// Waiting until the next 60 Hz tick would strand it for a whole interval.
|
||||||
|
aurora_begin_frame();
|
||||||
|
if (drawCount > 1 && frame >= warmupFrames && (frame + 1) % 60 == 0) {
|
||||||
|
std::printf("Completed %llu producer frames in %.2f s\n",
|
||||||
|
static_cast<unsigned long long>(frame + 1 - warmupFrames),
|
||||||
|
std::chrono::duration<double>(Clock::now() - measuredStart).count());
|
||||||
std::fflush(stdout);
|
std::fflush(stdout);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -176,10 +237,15 @@ int main(int argc, char** argv) {
|
|||||||
aurora_quiesce_frame_worker();
|
aurora_quiesce_frame_worker();
|
||||||
aurora_set_stereo_frame_provider(nullptr, nullptr);
|
aurora_set_stereo_frame_provider(nullptr, nullptr);
|
||||||
aurora::stereo::set_sink(nullptr, nullptr);
|
aurora::stereo::set_sink(nullptr, nullptr);
|
||||||
const double elapsed = std::chrono::duration<double>(Clock::now() - start).count();
|
const double elapsed = std::chrono::duration<double>(Clock::now() - measuredStart).count();
|
||||||
const double fps = submitted.load() / elapsed;
|
const auto measuredSubmissions = submitted.load() - initialSubmissions;
|
||||||
|
const double fps = measuredSubmissions / elapsed;
|
||||||
|
if (compositorFrames != 0)
|
||||||
|
std::printf("Compositor (including warm-up): wake late %.2f ms; submit %.2f ms; skipped %llu ticks\n",
|
||||||
|
wakeLateness / (1.0e6 * compositorFrames), submitTime / (1.0e6 * compositorFrames),
|
||||||
|
static_cast<unsigned long long>(skippedTicks));
|
||||||
std::printf("%u draws: producer %.1f FPS; %u stereo submissions in %.2f s (%.1f FPS)\n", drawCount, 240 / elapsed,
|
std::printf("%u draws: producer %.1f FPS; %u stereo submissions in %.2f s (%.1f FPS)\n", drawCount, 240 / elapsed,
|
||||||
submitted.load(), elapsed, fps);
|
measuredSubmissions, elapsed, fps);
|
||||||
aurora_shutdown();
|
aurora_shutdown();
|
||||||
// More headset submissions must not come at the expense of simulation speed.
|
// More headset submissions must not come at the expense of simulation speed.
|
||||||
const double target = mode == 2 ? 75 : mode == 1 ? 90 : 60;
|
const double target = mode == 2 ? 75 : mode == 1 ? 90 : 60;
|
||||||
|
|||||||
Reference in new issue
Block a user