diff --git a/OPENXR.md b/OPENXR.md
index 085093a..007205b 100644
--- a/OPENXR.md
+++ b/OPENXR.md
@@ -49,6 +49,7 @@ first_person_head_right_meters = 0.0
first_person_hide_driver = true
first_person_hidden_model = 0
first_person_rotation = "yaw"
+performance_level = "boost"
```
To play this installation on the desktop instead, set `enabled = false`, close the game completely,
@@ -127,6 +128,12 @@ desktop. `skip_copy_clears` independently suppresses the EFB reset performed aft
default on and can be changed live from the F10 settings bar for diagnostics.
`first_person` and the `first_person_*` values are the first-person camera described below. All
four are live and are also exposed in the F10 settings bar.
+`performance_level` is the level asked of the runtime through `XR_EXT_performance_settings` for
+its CPU and GPU domains: `boost`, `sustained_high`, `sustained_low`, `power_savings`, or
+`default` to leave the runtime's own choice. Standalone headsets clock their cores by this
+request (see `docs/quest-port.md`); desktop runtimes rarely offer the extension, and the setting
+then does nothing. It is read at launch, and the session log records whether the runtime accepted
+it and any later performance notification (a thermal or rendering warning).
## Controllers
@@ -475,8 +482,10 @@ Stereo uniform calculations use cached CPU memory, followed by a single write in
buffer. Reading or modifying matrices directly in D3D12 upload memory can be extremely slow,
especially with many character draws; see Microsoft's [Map guidance](https://learn.microsoft.com/en-us/windows/win32/api/d3d12/nf-d3d12-id3d12resource-map).
Retained interpolation reserves eye ranges at seal time and fills them once at the headset sample
-time. VR interpolation also releases the producer after sealing so eye encoding can overlap the
-next game frame, as it does with desktop interpolation.
+time. The frame worker always releases the producer after sealing, so eye encoding overlaps the
+next game frame whether or not interpolation is on. It used to publish that phase only after the
+encode unless interpolation was enabled, and the producer's first GX drain of every frame then
+waited for the previous frame's whole encode and submit (3 to 4.5 ms per frame on a Quest 3).
When VR interpolation is enabled at batch start, uniform recording also uses cached CPU
memory. Matching and history capture read that buffer, then the used prefix is copied to
diff --git a/android/app/src/main/java/org/wiicompiled/quest/launcher/SettingsPage.kt b/android/app/src/main/java/org/wiicompiled/quest/launcher/SettingsPage.kt
index 5c1cc4a..33a94aa 100644
--- a/android/app/src/main/java/org/wiicompiled/quest/launcher/SettingsPage.kt
+++ b/android/app/src/main/java/org/wiicompiled/quest/launcher/SettingsPage.kt
@@ -165,6 +165,16 @@ class SettingsPage(
format = { "%.2f×".format(it) },
write = { c, value -> c.setFloat("vr", "render_scale", value) },
)
+ choice(
+ R.string.vr_performance_level, R.string.vr_performance_level_helper,
+ listOf(
+ R.string.vr_performance_level_boost, R.string.vr_performance_level_sustained_high,
+ R.string.vr_performance_level_sustained_low, R.string.vr_performance_level_power_savings,
+ R.string.vr_performance_level_default,
+ ),
+ read = { stringIndex(it, "vr", "performance_level", PERFORMANCE_LEVELS) },
+ write = { c, index -> c.setString("vr", "performance_level", PERFORMANCE_LEVELS[index]) },
+ )
choice(
R.string.vr_interpolation, R.string.vr_interpolation_helper,
listOf(activity.getString(R.string.vr_interpolation_off), activity.getString(R.string.vr_interpolation_auto), "72 FPS", "90 FPS", "120 FPS"),
@@ -667,6 +677,8 @@ class SettingsPage(
private companion object {
val ROTATIONS = listOf("yaw", "yaw_pitch", "full")
+ // The runtime's default ("boost") first: an absent key reads as index 0.
+ val PERFORMANCE_LEVELS = listOf("boost", "sustained_high", "sustained_low", "power_savings", "default")
val CONTROLLER_MODES = listOf("wii_remote", "gamepad")
val INTERPOLATION_FPS = listOf(0L, 1L, 72L, 90L, 120L)
val RESOLUTIONS = listOf(1.0, 1.5, 2.0, 3.0, 4.0)
diff --git a/android/app/src/main/res/values/strings.xml b/android/app/src/main/res/values/strings.xml
index bd0a746..f651e54 100644
--- a/android/app/src/main/res/values/strings.xml
+++ b/android/app/src/main/res/values/strings.xml
@@ -157,6 +157,13 @@
HeadsetHeadset render scaleScales the eye resolution the headset recommends. Higher is sharper and costs GPU time; lower it if races stutter.
+ Headset performance level
+ How hard the headset clocks its CPU and GPU for the game. Boost gives the most speed; lower it if the headset gets hot or the battery matters more.
+ Boost
+ Sustained high
+ Sustained low
+ Power savings
+ Headset defaultVR frame interpolationExperimental. Renders extra frames between game frames at the headset\'s rate. Needs GPU headroom and adds one game frame of latency.Off
diff --git a/aurora-main/lib/aurora.cpp b/aurora-main/lib/aurora.cpp
index 1adef5c..b2f0a2b 100644
--- a/aurora-main/lib/aurora.cpp
+++ b/aurora-main/lib/aurora.cpp
@@ -443,9 +443,28 @@ bool frame_worker_phase_reached(FrameWorkerPhase phase) noexcept {
bool wait_for_frame_worker_private_for(FrameWorkerPhase phase, std::chrono::microseconds timeout) noexcept;
+// Where a frame's wall-clock time goes between the producer and the worker, summed in nanoseconds
+// and reported with the frame-rate log: the producer's waits for each worker phase, and the
+// worker cycle's four stretches: sealing, the wait for the producer's prepare permit, preparing
+// the next frame (which includes waiting for a free staging buffer), and the overlapped encode.
+std::atomic g_producerWaitDoneNs{0};
+std::atomic g_producerWaitSealedNs{0};
+std::atomic g_workerSealNs{0};
+std::atomic g_workerEncodeNs{0};
+std::atomic g_workerPermitWaitNs{0};
+std::atomic g_workerPrepareNs{0};
+
void wait_for_frame_worker_private(FrameWorkerPhase phase) noexcept {
constexpr auto kWaitServiceInterval = std::chrono::milliseconds(1);
+ if (frame_worker_phase_reached(phase)) {
+ return;
+ }
+ const auto started = std::chrono::steady_clock::now();
while (!wait_for_frame_worker_private_for(phase, kWaitServiceInterval)) {}
+ const auto waited = static_cast(
+ std::chrono::duration_cast(std::chrono::steady_clock::now() - started).count());
+ (phase == FrameWorkerPhase::Sealed ? g_producerWaitSealedNs : g_producerWaitDoneNs)
+ .fetch_add(waited, std::memory_order_relaxed);
}
bool wait_for_frame_worker_private_for(FrameWorkerPhase phase, std::chrono::microseconds timeout) noexcept {
@@ -2169,8 +2188,21 @@ void record_frame_telemetry() {
const auto now = std::chrono::steady_clock::now();
const std::chrono::duration elapsed = now - windowStart;
if (elapsed.count() >= 5.0) {
- Log.info("Game frame rate {:.1f} FPS ({} frames in {:.2f} s)", windowFrames / elapsed.count(), windowFrames,
- elapsed.count());
+ // Per-frame averages of where the wall-clock time went (see the counters' definition).
+ const auto msPerFrame = [&](std::atomic& counter) {
+ return static_cast(counter.exchange(0, std::memory_order_relaxed)) / 1e6 / std::max(windowFrames, 1u);
+ };
+ const double waitDone = msPerFrame(g_producerWaitDoneNs);
+ const double waitSealed = msPerFrame(g_producerWaitSealedNs);
+ const double seal = msPerFrame(g_workerSealNs);
+ const double permitWait = msPerFrame(g_workerPermitWaitNs);
+ const double prepare = msPerFrame(g_workerPrepareNs);
+ const double encode = msPerFrame(g_workerEncodeNs);
+ Log.info("Game frame rate {:.1f} FPS ({} frames in {:.2f} s); per frame the producer waited {:.2f} ms for "
+ "DONE and {:.2f} ms for SEALED; the worker spent {:.2f} ms sealing, {:.2f} ms waiting for the "
+ "prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding",
+ windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal,
+ permitWait, prepare, encode);
windowStart = now;
windowFrames = 0;
}
@@ -2187,19 +2219,17 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
ZoneScopedN("Frame worker cycle");
webgpu::fail_if_device_lost();
SealedFrameContext ctx;
- std::vector presentationJobs;
- bool overlapEncode = false;
+ const auto elapsedNs = [](std::chrono::steady_clock::time_point since) {
+ return static_cast(
+ std::chrono::duration_cast(std::chrono::steady_clock::now() - since).count());
+ };
+ auto stretchStarted = std::chrono::steady_clock::now();
{
std::lock_guard gpuLock(g_rendererGpuMutex);
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
- overlapEncode = ctx.interpolationActive || ctx.retainStereo;
- if (!overlapEncode) {
- presentationJobs = encode_sealed_frame(sealedFrame, ctx);
- }
- }
- if (!overlapEncode) {
- publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
}
+ g_workerSealNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
+ stretchStarted = std::chrono::steady_clock::now();
{
std::unique_lock lock(g_frameWorker.mutex);
@@ -2209,39 +2239,41 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
}
g_frameWorker.prepareAllowed = false;
}
+ g_workerPermitWaitNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
+ stretchStarted = std::chrono::steady_clock::now();
// Preparing the next frame belongs to the SEALED phase: without a fresh pass 0 and mapped
// staging buffers the producer's drain has nowhere to put its commands.
bool imguiNewFrameOwed = false;
- const bool prepared = begin_frame_impl(
- false, overlapEncode ? ImGuiFramePolicy::Deferred : ImGuiFramePolicy::Immediate, &imguiNewFrameOwed);
+ const bool prepared = begin_frame_impl(false, ImGuiFramePolicy::Deferred, &imguiNewFrameOwed);
+ g_workerPrepareNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
+ stretchStarted = std::chrono::steady_clock::now();
+ // SEALED before the encode, always. The encode reads only `ctx` and the sealed passes, so it
+ // runs while the producer records the next frame, which takes the renderer mutex per drain.
+ // It used to be published after the encode unless interpolation was on, and the producer's
+ // first drain of each frame then waited for the whole encode and submit: 3 to 4.5 ms of every
+ // frame on a Quest 3, the difference between a twelve-kart race start at 50 and at 60 fps.
{
std::lock_guard lock(g_frameWorker.mutex);
g_frameWorker.framePrepared = prepared;
g_frameWorker.sealed.store(true, std::memory_order_release);
- if (!overlapEncode) {
- g_frameWorker.ready.store(true, std::memory_order_release);
- }
}
g_frameWorker.cv.notify_all();
- if (overlapEncode) {
- // Mutex-free: the producer drains and records the next frame in parallel, taking the renderer
- // mutex per drain, and this phase never takes it.
- presentationJobs = encode_sealed_frame(sealedFrame, ctx);
- publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
- if (imguiNewFrameOwed) {
- // Safe only now: every slot has replayed this frame's ImGui draw lists.
- std::lock_guard gpuLock(g_rendererGpuMutex);
- imgui::new_frame(window::get_window_size());
- }
- {
- std::lock_guard lock(g_frameWorker.mutex);
- g_frameWorker.ready.store(true, std::memory_order_release);
- }
- g_frameWorker.cv.notify_all();
+ std::vector presentationJobs = encode_sealed_frame(sealedFrame, ctx);
+ publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
+ if (imguiNewFrameOwed) {
+ // Safe only now: every slot has replayed this frame's ImGui draw lists.
+ std::lock_guard gpuLock(g_rendererGpuMutex);
+ imgui::new_frame(window::get_window_size());
}
+ g_workerEncodeNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
+ {
+ std::lock_guard lock(g_frameWorker.mutex);
+ g_frameWorker.ready.store(true, std::memory_order_release);
+ }
+ g_frameWorker.cv.notify_all();
record_frame_telemetry();
return true;
diff --git a/aurora-main/lib/gx/command_processor.cpp b/aurora-main/lib/gx/command_processor.cpp
index 7fb3096..3f38b5b 100644
--- a/aurora-main/lib/gx/command_processor.cpp
+++ b/aurora-main/lib/gx/command_processor.cpp
@@ -1833,16 +1833,44 @@ static void handle_draw_overrun(u8 cmd, u16 vtxCount, u32 vtxSize, u32 totalVtxB
// Uploads a draw's GX vertices with the stride populate_pipeline_config gave the shader. When that stride is padded
// (Android, see padded_upload_stride), each vertex is copied with zeroed trailing bytes; attribute offsets inside the
// vertex are unchanged. Every padded upload is a multiple of 4 bytes, so consecutive draws stay contiguous for merging.
+// One padded element: srcStride bytes copied in word, half-word and byte steps, then the
+// padding zeroed. Inline arithmetic instead of a memcpy and memset call per element: a race
+// frame uploads tens of thousands of 7-byte vertices and 6-byte normals this way, and the two
+// library calls per element were most of the game thread's memcpy time on the Quest.
+static inline void copy_padded_element(u8* dst, const u8* src, u32 srcStride, u32 dstStride) noexcept {
+ u32 n = srcStride;
+ while (n >= 4) {
+ u32 word;
+ std::memcpy(&word, src, 4);
+ std::memcpy(dst, &word, 4);
+ src += 4;
+ dst += 4;
+ n -= 4;
+ }
+ if (n >= 2) {
+ u16 half;
+ std::memcpy(&half, src, 2);
+ std::memcpy(dst, &half, 2);
+ src += 2;
+ dst += 2;
+ n -= 2;
+ }
+ if (n != 0) {
+ *dst++ = *src;
+ }
+ for (u32 pad = srcStride; pad < dstStride; ++pad) {
+ *dst++ = 0;
+ }
+}
+
static gfx::Range push_draw_vertices(const u8* vertices, u32 vtxCount, u32 vtxSize) {
const u32 uploadStride = padded_upload_stride(vtxSize);
if (uploadStride == vtxSize)
LIKELY { return gfx::push_verts(vertices, static_cast(vtxCount) * vtxSize); }
auto [buffer, range] = gfx::map_verts(static_cast(vtxCount) * uploadStride);
u8* dst = buffer.data();
- const u32 padding = uploadStride - vtxSize;
for (u32 i = 0; i < vtxCount; ++i) {
- std::memcpy(dst, vertices + static_cast(i) * vtxSize, vtxSize);
- std::memset(dst + vtxSize, 0, padding);
+ copy_padded_element(dst, vertices + static_cast(i) * vtxSize, vtxSize, uploadStride);
dst += uploadStride;
}
return range;
@@ -1857,10 +1885,15 @@ static gfx::Range push_vertex_array(const AttrArray& array, u32 uploadStride) {
const size_t count = (static_cast(array.size) + array.stride - 1) / array.stride;
auto [buffer, range] = gfx::map_storage(count * uploadStride);
u8* dst = buffer.data();
- std::memset(dst, 0, count * uploadStride);
- for (size_t i = 0; i < count; ++i) {
- const size_t start = i * array.stride;
- std::memcpy(dst + i * uploadStride, data + start, std::min(array.stride, array.size - start));
+ // Every whole element inline; only a trailing partial element takes the library calls.
+ const size_t whole = static_cast(array.size) / array.stride;
+ for (size_t i = 0; i < whole; ++i) {
+ copy_padded_element(dst + i * uploadStride, data + i * array.stride, array.stride, uploadStride);
+ }
+ if (whole < count) {
+ const size_t start = whole * array.stride;
+ std::memset(dst + whole * uploadStride, 0, uploadStride);
+ std::memcpy(dst + whole * uploadStride, data + start, array.size - start);
}
return range;
}
diff --git a/docs/quest-port.md b/docs/quest-port.md
index fccd8c6..4c556c2 100644
--- a/docs/quest-port.md
+++ b/docs/quest-port.md
@@ -582,7 +582,7 @@ the app:
| `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update |
| `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds |
| `debug.wiicompiled.inject :