Added performance level configuration for OpenXR runtime and enhance Aurora's frame worker

This commit is contained in:
iChris4 committed 2026-09-19 04:43:47 +02:00
1 parent d296b3dcb3
commit 60c443654f
12 files changed
+292 -54

No files matched your search

+11 -2
View File
@@ -49,6 +49,7 @@ first_person_head_right_meters = 0.0
first_person_hide_driver = true
first_person_hidden_model = 0
first_person_rotation = "yaw"
performance_level = "boost"
```
To play this installation on the desktop instead, set `enabled = false`, close the game completely,
@@ -127,6 +128,12 @@ desktop. `skip_copy_clears` independently suppresses the EFB reset performed aft
default on and can be changed live from the F10 settings bar for diagnostics.
`first_person` and the `first_person_*` values are the first-person camera described below. All
four are live and are also exposed in the F10 settings bar.
`performance_level` is the level asked of the runtime through `XR_EXT_performance_settings` for
its CPU and GPU domains: `boost`, `sustained_high`, `sustained_low`, `power_savings`, or
`default` to leave the runtime's own choice. Standalone headsets clock their cores by this
request (see `docs/quest-port.md`); desktop runtimes rarely offer the extension, and the setting
then does nothing. It is read at launch, and the session log records whether the runtime accepted
it and any later performance notification (a thermal or rendering warning).
## Controllers
@@ -475,8 +482,10 @@ Stereo uniform calculations use cached CPU memory, followed by a single write in
buffer. Reading or modifying matrices directly in D3D12 upload memory can be extremely slow,
especially with many character draws; see Microsoft's [Map guidance](https://learn.microsoft.com/en-us/windows/win32/api/d3d12/nf-d3d12-id3d12resource-map).
Retained interpolation reserves eye ranges at seal time and fills them once at the headset sample
time. VR interpolation also releases the producer after sealing so eye encoding can overlap the
next game frame, as it does with desktop interpolation.
time. The frame worker always releases the producer after sealing, so eye encoding overlaps the
next game frame whether or not interpolation is on. It used to publish that phase only after the
encode unless interpolation was enabled, and the producer's first GX drain of every frame then
waited for the previous frame's whole encode and submit (3 to 4.5 ms per frame on a Quest 3).
When VR interpolation is enabled at batch start, uniform recording also uses cached CPU
memory. Matching and history capture read that buffer, then the used prefix is copied to
@@ -165,6 +165,16 @@ class SettingsPage(
format = { "%.2f×".format(it) },
write = { c, value -> c.setFloat("vr", "render_scale", value) },
)
choice(
R.string.vr_performance_level, R.string.vr_performance_level_helper,
listOf(
R.string.vr_performance_level_boost, R.string.vr_performance_level_sustained_high,
R.string.vr_performance_level_sustained_low, R.string.vr_performance_level_power_savings,
R.string.vr_performance_level_default,
),
read = { stringIndex(it, "vr", "performance_level", PERFORMANCE_LEVELS) },
write = { c, index -> c.setString("vr", "performance_level", PERFORMANCE_LEVELS[index]) },
)
choice(
R.string.vr_interpolation, R.string.vr_interpolation_helper,
listOf(activity.getString(R.string.vr_interpolation_off), activity.getString(R.string.vr_interpolation_auto), "72 FPS", "90 FPS", "120 FPS"),
@@ -667,6 +677,8 @@ class SettingsPage(
private companion object {
val ROTATIONS = listOf("yaw", "yaw_pitch", "full")
// The runtime's default ("boost") first: an absent key reads as index 0.
val PERFORMANCE_LEVELS = listOf("boost", "sustained_high", "sustained_low", "power_savings", "default")
val CONTROLLER_MODES = listOf("wii_remote", "gamepad")
val INTERPOLATION_FPS = listOf(0L, 1L, 72L, 90L, 120L)
val RESOLUTIONS = listOf(1.0, 1.5, 2.0, 3.0, 4.0)
@@ -157,6 +157,13 @@
<string name="section_vr_headset">Headset</string>
<string name="vr_render_scale">Headset render scale</string>
<string name="vr_render_scale_helper">Scales the eye resolution the headset recommends. Higher is sharper and costs GPU time; lower it if races stutter.</string>
<string name="vr_performance_level">Headset performance level</string>
<string name="vr_performance_level_helper">How hard the headset clocks its CPU and GPU for the game. Boost gives the most speed; lower it if the headset gets hot or the battery matters more.</string>
<string name="vr_performance_level_boost">Boost</string>
<string name="vr_performance_level_sustained_high">Sustained high</string>
<string name="vr_performance_level_sustained_low">Sustained low</string>
<string name="vr_performance_level_power_savings">Power savings</string>
<string name="vr_performance_level_default">Headset default</string>
<string name="vr_interpolation">VR frame interpolation</string>
<string name="vr_interpolation_helper">Experimental. Renders extra frames between game frames at the headset\'s rate. Needs GPU headroom and adds one game frame of latency.</string>
<string name="vr_interpolation_off">Off</string>
+63 -31
View File
@@ -443,9 +443,28 @@ bool frame_worker_phase_reached(FrameWorkerPhase phase) noexcept {
bool wait_for_frame_worker_private_for(FrameWorkerPhase phase, std::chrono::microseconds timeout) noexcept;
// Where a frame's wall-clock time goes between the producer and the worker, summed in nanoseconds
// and reported with the frame-rate log: the producer's waits for each worker phase, and the
// worker cycle's four stretches: sealing, the wait for the producer's prepare permit, preparing
// the next frame (which includes waiting for a free staging buffer), and the overlapped encode.
std::atomic<uint64_t> g_producerWaitDoneNs{0};
std::atomic<uint64_t> g_producerWaitSealedNs{0};
std::atomic<uint64_t> g_workerSealNs{0};
std::atomic<uint64_t> g_workerEncodeNs{0};
std::atomic<uint64_t> g_workerPermitWaitNs{0};
std::atomic<uint64_t> g_workerPrepareNs{0};
void wait_for_frame_worker_private(FrameWorkerPhase phase) noexcept {
constexpr auto kWaitServiceInterval = std::chrono::milliseconds(1);
if (frame_worker_phase_reached(phase)) {
return;
}
const auto started = std::chrono::steady_clock::now();
while (!wait_for_frame_worker_private_for(phase, kWaitServiceInterval)) {}
const auto waited = static_cast<uint64_t>(
std::chrono::duration_cast<std::chrono::nanoseconds>(std::chrono::steady_clock::now() - started).count());
(phase == FrameWorkerPhase::Sealed ? g_producerWaitSealedNs : g_producerWaitDoneNs)
.fetch_add(waited, std::memory_order_relaxed);
}
bool wait_for_frame_worker_private_for(FrameWorkerPhase phase, std::chrono::microseconds timeout) noexcept {
@@ -2169,8 +2188,21 @@ void record_frame_telemetry() {
const auto now = std::chrono::steady_clock::now();
const std::chrono::duration<double> elapsed = now - windowStart;
if (elapsed.count() >= 5.0) {
Log.info("Game frame rate {:.1f} FPS ({} frames in {:.2f} s)", windowFrames / elapsed.count(), windowFrames,
elapsed.count());
// Per-frame averages of where the wall-clock time went (see the counters' definition).
const auto msPerFrame = [&](std::atomic<uint64_t>& counter) {
return static_cast<double>(counter.exchange(0, std::memory_order_relaxed)) / 1e6 / std::max(windowFrames, 1u);
};
const double waitDone = msPerFrame(g_producerWaitDoneNs);
const double waitSealed = msPerFrame(g_producerWaitSealedNs);
const double seal = msPerFrame(g_workerSealNs);
const double permitWait = msPerFrame(g_workerPermitWaitNs);
const double prepare = msPerFrame(g_workerPrepareNs);
const double encode = msPerFrame(g_workerEncodeNs);
Log.info("Game frame rate {:.1f} FPS ({} frames in {:.2f} s); per frame the producer waited {:.2f} ms for "
"DONE and {:.2f} ms for SEALED; the worker spent {:.2f} ms sealing, {:.2f} ms waiting for the "
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding",
windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal,
permitWait, prepare, encode);
windowStart = now;
windowFrames = 0;
}
@@ -2187,19 +2219,17 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
ZoneScopedN("Frame worker cycle");
webgpu::fail_if_device_lost();
SealedFrameContext ctx;
std::vector<PresentationJob> presentationJobs;
bool overlapEncode = false;
const auto elapsedNs = [](std::chrono::steady_clock::time_point since) {
return static_cast<uint64_t>(
std::chrono::duration_cast<std::chrono::nanoseconds>(std::chrono::steady_clock::now() - since).count());
};
auto stretchStarted = std::chrono::steady_clock::now();
{
std::lock_guard gpuLock(g_rendererGpuMutex);
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
overlapEncode = ctx.interpolationActive || ctx.retainStereo;
if (!overlapEncode) {
presentationJobs = encode_sealed_frame(sealedFrame, ctx);
}
}
if (!overlapEncode) {
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
}
g_workerSealNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
stretchStarted = std::chrono::steady_clock::now();
{
std::unique_lock lock(g_frameWorker.mutex);
@@ -2209,39 +2239,41 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
}
g_frameWorker.prepareAllowed = false;
}
g_workerPermitWaitNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
stretchStarted = std::chrono::steady_clock::now();
// Preparing the next frame belongs to the SEALED phase: without a fresh pass 0 and mapped
// staging buffers the producer's drain has nowhere to put its commands.
bool imguiNewFrameOwed = false;
const bool prepared = begin_frame_impl(
false, overlapEncode ? ImGuiFramePolicy::Deferred : ImGuiFramePolicy::Immediate, &imguiNewFrameOwed);
const bool prepared = begin_frame_impl(false, ImGuiFramePolicy::Deferred, &imguiNewFrameOwed);
g_workerPrepareNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
stretchStarted = std::chrono::steady_clock::now();
// SEALED before the encode, always. The encode reads only `ctx` and the sealed passes, so it
// runs while the producer records the next frame, which takes the renderer mutex per drain.
// It used to be published after the encode unless interpolation was on, and the producer's
// first drain of each frame then waited for the whole encode and submit: 3 to 4.5 ms of every
// frame on a Quest 3, the difference between a twelve-kart race start at 50 and at 60 fps.
{
std::lock_guard lock(g_frameWorker.mutex);
g_frameWorker.framePrepared = prepared;
g_frameWorker.sealed.store(true, std::memory_order_release);
if (!overlapEncode) {
g_frameWorker.ready.store(true, std::memory_order_release);
}
}
g_frameWorker.cv.notify_all();
if (overlapEncode) {
// Mutex-free: the producer drains and records the next frame in parallel, taking the renderer
// mutex per drain, and this phase never takes it.
presentationJobs = encode_sealed_frame(sealedFrame, ctx);
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
if (imguiNewFrameOwed) {
// Safe only now: every slot has replayed this frame's ImGui draw lists.
std::lock_guard gpuLock(g_rendererGpuMutex);
imgui::new_frame(window::get_window_size());
}
{
std::lock_guard lock(g_frameWorker.mutex);
g_frameWorker.ready.store(true, std::memory_order_release);
}
g_frameWorker.cv.notify_all();
std::vector<PresentationJob> presentationJobs = encode_sealed_frame(sealedFrame, ctx);
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
if (imguiNewFrameOwed) {
// Safe only now: every slot has replayed this frame's ImGui draw lists.
std::lock_guard gpuLock(g_rendererGpuMutex);
imgui::new_frame(window::get_window_size());
}
g_workerEncodeNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
{
std::lock_guard lock(g_frameWorker.mutex);
g_frameWorker.ready.store(true, std::memory_order_release);
}
g_frameWorker.cv.notify_all();
record_frame_telemetry();
return true;
+40 -7
View File
@@ -1833,16 +1833,44 @@ static void handle_draw_overrun(u8 cmd, u16 vtxCount, u32 vtxSize, u32 totalVtxB
// Uploads a draw's GX vertices with the stride populate_pipeline_config gave the shader. When that stride is padded
// (Android, see padded_upload_stride), each vertex is copied with zeroed trailing bytes; attribute offsets inside the
// vertex are unchanged. Every padded upload is a multiple of 4 bytes, so consecutive draws stay contiguous for merging.
// One padded element: srcStride bytes copied in word, half-word and byte steps, then the
// padding zeroed. Inline arithmetic instead of a memcpy and memset call per element: a race
// frame uploads tens of thousands of 7-byte vertices and 6-byte normals this way, and the two
// library calls per element were most of the game thread's memcpy time on the Quest.
static inline void copy_padded_element(u8* dst, const u8* src, u32 srcStride, u32 dstStride) noexcept {
u32 n = srcStride;
while (n >= 4) {
u32 word;
std::memcpy(&word, src, 4);
std::memcpy(dst, &word, 4);
src += 4;
dst += 4;
n -= 4;
}
if (n >= 2) {
u16 half;
std::memcpy(&half, src, 2);
std::memcpy(dst, &half, 2);
src += 2;
dst += 2;
n -= 2;
}
if (n != 0) {
*dst++ = *src;
}
for (u32 pad = srcStride; pad < dstStride; ++pad) {
*dst++ = 0;
}
}
static gfx::Range push_draw_vertices(const u8* vertices, u32 vtxCount, u32 vtxSize) {
const u32 uploadStride = padded_upload_stride(vtxSize);
if (uploadStride == vtxSize)
LIKELY { return gfx::push_verts(vertices, static_cast<size_t>(vtxCount) * vtxSize); }
auto [buffer, range] = gfx::map_verts(static_cast<size_t>(vtxCount) * uploadStride);
u8* dst = buffer.data();
const u32 padding = uploadStride - vtxSize;
for (u32 i = 0; i < vtxCount; ++i) {
std::memcpy(dst, vertices + static_cast<size_t>(i) * vtxSize, vtxSize);
std::memset(dst + vtxSize, 0, padding);
copy_padded_element(dst, vertices + static_cast<size_t>(i) * vtxSize, vtxSize, uploadStride);
dst += uploadStride;
}
return range;
@@ -1857,10 +1885,15 @@ static gfx::Range push_vertex_array(const AttrArray& array, u32 uploadStride) {
const size_t count = (static_cast<size_t>(array.size) + array.stride - 1) / array.stride;
auto [buffer, range] = gfx::map_storage(count * uploadStride);
u8* dst = buffer.data();
std::memset(dst, 0, count * uploadStride);
for (size_t i = 0; i < count; ++i) {
const size_t start = i * array.stride;
std::memcpy(dst + i * uploadStride, data + start, std::min<size_t>(array.stride, array.size - start));
// Every whole element inline; only a trailing partial element takes the library calls.
const size_t whole = static_cast<size_t>(array.size) / array.stride;
for (size_t i = 0; i < whole; ++i) {
copy_padded_element(dst + i * uploadStride, data + i * array.stride, array.stride, uploadStride);
}
if (whole < count) {
const size_t start = whole * array.stride;
std::memset(dst + whole * uploadStride, 0, uploadStride);
std::memcpy(dst + whole * uploadStride, data + start, array.size - start);
}
return range;
}
+28 -1
View File
@@ -582,7 +582,7 @@ the app:
| `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update |
| `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds |
| `debug.wiicompiled.inject <n>:<button>` | Presses `a`, `b`, `x`, `y`, `start`, `up`, `down`, `left` or `right` for 12 XR frames each time `<n>` changes. As a Wii Remote, `x`/`y`/`start` are 1/2/+, the directions push the Nunchuk stick, and `home`, `c` and `z` also exist. `panel` clicks both thumbsticks, opening or closing the settings panel (see `OPENXR.md`) |
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s. The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%` and app GPU time (`App=`) |
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
The injector makes headset tests possible with nobody wearing the headset.
Keep the display awake, drive the menus, then take a compositor screenshot:
@@ -622,6 +622,33 @@ follows the game rate instead of holding 72 Hz, so the pacing thread is not
repeating the last layer as it does on desktop. Both need profiling on the
XR2 Gen 2.
Profiled 2026-09-19 on a twelve-kart 50cc Grand Prix start (Luigi Circuit,
`render_scale` 0.8, intro skipped, driven unattended by the button injector:
ten `a` presses 5 s apart from the title screen reach the race, one more skips
the course intro). The game thread is the limit, not the GPU: it is one
libco-hosted thread (about 60% translated game code, 20% GX HLE, 12% Aurora's
FIFO decode), and Aurora's frame worker, which encodes and submits the Dawn
work, runs at about half a core with most of its own time inside the Adreno
driver's ioctls. Trimming the GX HLE (a 4 KiB write-tracking granule, one
pointer probe for the texture-object shadow, inline padded vertex copies, a
throttled clock poll) changed nothing measurable against the same automated
start, and asking for `XR_EXT_performance_settings` BOOST is accepted but the
runtime keeps its own dynamic clocks (CPU level 4 at 1.9 to 2.2 GHz, GPU level
3 at 490 to 640 MHz). What did matter was a scheduler trace of the game
thread: it slept 3 to 4.5 ms of every frame, in 1 ms slices, on Aurora's
SEALED phase. Without interpolation the worker published SEALED only after the
whole encode and submit, so the producer's first GX drain of each frame waited
for the previous frame's encode (5 ms in menus, 9 to 11 ms in a race). The
worker now always releases the producer right after sealing; the same start
went from 47 to 53 fps to 55 to 58 fps, the game thread from 82% to 98% busy,
and menus lost the same 4.5 ms of idle wait per frame. With `fpslog` on, the
`Game frame rate` line now carries that breakdown (producer waits for DONE and
SEALED; the worker's seal, permit wait, prepare and encode) so the next
regression of this kind shows up in the session log. The remaining gap to 60
at the start is about 1 ms of game-thread CPU per frame, with the GPU at 85 to
89%, so the next steps are on both sides: the guest-code share (translator
output quality) and the eye replay's GPU cost.
Verified on device since: the menus on the virtual screen, controller input
(the user has driven races), and an immersive Grand Prix start with all 12
racers rendering correctly. Not yet verified: stereo comfort and scale,
+6 -4
View File
@@ -39,10 +39,12 @@ inline uint32_t CanonicalizeGxMainRamAddress(uint32_t addr) noexcept {
namespace GxGuestWrite {
// Granularity matches the GX resource caches' occupancy maps: 64 KiB, so a
// cacheable display list or texture spans only a handful of counters. A false
// bump only costs one re-digest, which is exactly the untracked behaviour.
inline constexpr uint32_t kGranuleShift = 16; // 64 KiB per granule
// 4 KiB per granule, so a cacheable display list or texture spans a few counters and a false
// bump only costs one re-digest, which is exactly the untracked behaviour. It used to be 64 KiB:
// then per-frame guest writes landing beside a static list or texture (a race start streams
// data next to them) bumped the same granule, and every affected list was re-digested and every
// affected texture re-hashed each frame. The table is 320 KiB.
inline constexpr uint32_t kGranuleShift = 12;
inline constexpr uint64_t kTrackedSpan = static_cast<uint64_t>(Memory::kMem2PhysicalEnd);
inline constexpr size_t kGranuleCount = static_cast<size_t>(kTrackedSpan >> kGranuleShift);
+33
View File
@@ -68,6 +68,7 @@ struct RuntimeUserConfig {
std::optional<bool> vrFirstPersonHideDriver;
std::optional<int32_t> vrFirstPersonHiddenModel;
std::optional<std::string> vrFirstPersonRotation;
std::optional<std::string> vrPerformanceLevel;
std::optional<std::string> vrRecenterKey;
std::optional<float> vrLeanBackDegrees;
// F10 > Diagnostics: OpenXR pacing and presentation logging in console.log.
@@ -179,6 +180,16 @@ inline constexpr const char* kVrFirstPersonRotationDefault = "yaw";
inline bool IsSupportedVrFirstPersonRotation(std::string_view value) {
return value == "yaw" || value == "yaw_pitch" || value == "full";
}
// The performance level asked of the OpenXR runtime (XR_EXT_performance_settings) for its CPU and
// GPU domains. Standalone headsets clock their cores by this request: a Quest 3 ran the game
// thread at 1.92 GHz with the runtime's own choice while its fast cores reach 2.36 GHz. "default"
// leaves the runtime's choice; desktop runtimes without the extension ignore the setting.
inline constexpr const char* kVrPerformanceLevelDefault = "boost";
inline bool IsSupportedVrPerformanceLevel(std::string_view value) {
return value == "default" || value == "power_savings" || value == "sustained_low" ||
value == "sustained_high" || value == "boost";
}
// What the desktop window shows while the headset is running: "normal" leaves
// the ordinary desktop view alone, "both", "left" and "right" mirror the
// headset's eyes, and "none" blacks the window out. Matches
@@ -455,6 +466,11 @@ inline void EnsureConfigFile() {
"# horizon, \"yaw_pitch\" adds the kart's climb but no roll, and\n"
"# \"full\" takes the kart's whole orientation so the view banks.\n"
"first_person_rotation = \"yaw\"\n\n"
"# Performance level asked of the headset's runtime for its CPU and\n"
"# GPU: \"boost\", \"sustained_high\", \"sustained_low\", \"power_savings\",\n"
"# or \"default\" to leave the runtime's own choice. Standalone headsets\n"
"# clock their cores by this request; desktop runtimes ignore it.\n"
"performance_level = \"boost\"\n\n"
"# Keyboard shortcut that recenters the VR view, naming the key the\n"
"# way SDL does (F9, Home, Keypad 5, ...). It moves the race view to\n"
"# where you are sitting now and brings the menu screen back upright in\n"
@@ -675,6 +691,10 @@ inline RuntimeUserConfig ParseConfigDocument(const toml::value& document) {
value && IsSupportedVrFirstPersonRotation(*value)) {
config.vrFirstPersonRotation = *value;
}
if (auto value = FindConfigValue<std::string>(document, "vr", "performance_level");
value && IsSupportedVrPerformanceLevel(*value)) {
config.vrPerformanceLevel = *value;
}
if (auto value = FindConfigValue<std::string>(document, "vr", "mirror_view");
value && IsSupportedVrMirrorView(*value)) {
config.vrMirrorView = *value;
@@ -1013,6 +1033,14 @@ inline bool SetVrFirstPersonRotation(std::string value) {
return WriteSetting("vr", "first_person_rotation", FormatString(value));
}
inline bool SetVrPerformanceLevel(std::string value) {
if (!IsSupportedVrPerformanceLevel(value)) {
return false;
}
Mutable().vrPerformanceLevel = value;
return WriteSetting("vr", "performance_level", FormatString(value));
}
inline bool SetVrFirstPersonHiddenModel(int32_t value) {
value = std::clamp(value, -1, 31);
Mutable().vrFirstPersonHiddenModel = value;
@@ -1371,6 +1399,11 @@ inline std::string VrFirstPersonRotation(std::string fallback = kVrFirstPersonRo
return value && IsSupportedVrFirstPersonRotation(*value) ? *value : std::move(fallback);
}
inline std::string VrPerformanceLevel(std::string fallback = kVrPerformanceLevelDefault) {
const auto& value = Get().vrPerformanceLevel;
return value && IsSupportedVrPerformanceLevel(*value) ? *value : std::move(fallback);
}
inline int32_t VrFirstPersonHiddenModel(int32_t fallback = kVrFirstPersonHiddenModelDefault) {
return std::clamp(Get().vrFirstPersonHiddenModel.value_or(fallback), -1, 31);
}
+16 -7
View File
@@ -2,6 +2,7 @@
#include "runtime_log.h"
#include <cstddef>
#include <cstring>
#include <limits>
// --- Texture and TLUT Objects ---
@@ -447,15 +448,23 @@ TexObjMeta MergeGuestTexObjMeta(const TexObjMeta& cached, const TexObjMeta& gues
// Reads the raw 32 guest bytes of a GXTexObj struct. Returns false (and the
// caller must treat the shadow as absent) when the address is unreadable.
// The shadow is only ever compared for equality (ShadowEquals), so it keeps
// the bytes in guest order: one page-table probe and one copy per lookup
// instead of four checked, byte-swapped 64-bit reads. Every draw's texture
// binds go through this on the game thread, so the per-call cost matters.
bool ReadTexObjShadow(uint32_t addr, uint64_t (&out)[4]) noexcept {
try {
out[0] = Memory::Read64(addr + 0x00);
out[1] = Memory::Read64(addr + 0x08);
out[2] = Memory::Read64(addr + 0x10);
out[3] = Memory::Read64(addr + 0x18);
} catch (...) {
return false;
const uint8_t* bytes = MemoryInline::GetPointerFast(addr, sizeof(out));
if (bytes == nullptr) {
try {
bytes = Memory::GetPointer(addr, sizeof(out));
} catch (...) {
return false;
}
if (bytes == nullptr) {
return false;
}
}
std::memcpy(out, bytes, sizeof(out));
return true;
}
+10
View File
@@ -244,6 +244,16 @@ void ServiceDeferredTimingDuringGxWork() {
return;
}
// GX__Begin runs thousands of times a frame, and reading the clock on each one was 2% of the
// game thread on the Quest. The poll has a 1 ms cadence, so sampling the clock on every
// sixteenth call keeps it within a few microseconds of that. A plain static: GX runs on the
// one guest-facing thread, and a thread_local here would cost a resolver call per access.
static uint32_t s_pollCountdown = 0;
if (s_pollCountdown != 0) {
--s_pollCountdown;
return;
}
s_pollCountdown = 15;
const auto now = std::chrono::steady_clock::now();
if (now < g_nextDeferredTimingPoll) {
return;
+34 -2
View File
@@ -369,7 +369,7 @@ public:
#if defined(_WIN32)
config.required_extensions = {"XR_KHR_D3D12_enable"};
config.optional_extensions = {"XR_KHR_win32_convert_performance_counter_time",
"XR_FB_display_refresh_rate"};
"XR_FB_display_refresh_rate", "XR_EXT_performance_settings"};
#else
// Either Vulkan binding extension is acceptable; the backend picks
// whichever the runtime enabled, preferring enable2.
@@ -377,7 +377,7 @@ public:
config.optional_extensions = {"XR_KHR_vulkan_enable2", "XR_KHR_vulkan_enable",
"XR_KHR_convert_timespec_time",
"XR_KHR_android_thread_settings",
"XR_FB_display_refresh_rate"};
"XR_FB_display_refresh_rate", "XR_EXT_performance_settings"};
config.instance_create_next = OpenXRAndroidInstanceCreateNext();
#endif
if (!runtime_->Initialize(config)) {
@@ -401,6 +401,9 @@ public:
if (has_extension("XR_FB_display_refresh_rate")) {
runtime_->LoadFunction("xrGetDisplayRefreshRateFB", &get_display_refresh_rate_);
}
if (has_extension("XR_EXT_performance_settings")) {
runtime_->LoadFunction("xrPerfSettingsSetPerformanceLevelEXT", &set_performance_level_);
}
interpolation_available_.store(convert_display_time_ != nullptr, std::memory_order_release);
if (!backend_->QueryGraphicsRequirements(*runtime_)) {
SetError(backend_->LastError());
@@ -509,6 +512,7 @@ public:
prepared_ = false;
convert_display_time_ = nullptr;
get_display_refresh_rate_ = nullptr;
set_performance_level_ = nullptr;
headset_hz_.store(0, std::memory_order_relaxed);
rendered_fps_.store(0, std::memory_order_relaxed);
interpolation_available_.store(false, std::memory_order_release);
@@ -635,6 +639,32 @@ private:
}
#endif
// Asks the runtime for the configured performance level in both domains. Standalone
// headsets clock their cores by this: a Quest 3 held the game thread at CPU level 4
// (2.2 GHz of a possible 2.36) and the GPU at level 3 with the runtime's own choice. A
// refusal is logged and changes nothing; desktop runtimes rarely offer the extension.
void ApplyPerformanceLevel() {
if (runtime_ == nullptr || set_performance_level_ == nullptr || !runtime_->HasSession()) {
return;
}
const std::string requested = RuntimeConfigFile::VrPerformanceLevel();
XrPerfSettingsLevelEXT level = XR_PERF_SETTINGS_LEVEL_SUSTAINED_HIGH_EXT;
if (requested == "default") {
return;
} else if (requested == "power_savings") {
level = XR_PERF_SETTINGS_LEVEL_POWER_SAVINGS_EXT;
} else if (requested == "sustained_low") {
level = XR_PERF_SETTINGS_LEVEL_SUSTAINED_LOW_EXT;
} else if (requested == "boost") {
level = XR_PERF_SETTINGS_LEVEL_BOOST_EXT;
}
const XrResult cpu = set_performance_level_(runtime_->Session(), XR_PERF_SETTINGS_DOMAIN_CPU_EXT, level);
const XrResult gpu = set_performance_level_(runtime_->Session(), XR_PERF_SETTINGS_DOMAIN_GPU_EXT, level);
RT_LOG(RT_TAG_RUNTIME) << "OpenXR: performance level \"" << requested << "\" CPU "
<< (XR_SUCCEEDED(cpu) ? "set" : "refused") << " (" << cpu << "), GPU "
<< (XR_SUCCEEDED(gpu) ? "set" : "refused") << " (" << gpu << ")" << std::endl;
}
static bool ProvideStereoFrame(uint32_t, AuroraStereoFrame* output, void* userdata) {
auto* self = static_cast<OpenXRIntegration*>(userdata);
if (self == nullptr || output == nullptr) {
@@ -672,6 +702,7 @@ private:
worker_registered = RegisterAuroraFrameWorkerThread();
}
#endif
ApplyPerformanceLevel();
bool fatal = false;
uint32_t consecutive_skips = 0;
bool store_gate_set = false;
@@ -1295,6 +1326,7 @@ private:
std::chrono::steady_clock::time_point timing_start_ = std::chrono::steady_clock::now();
uint32_t timing_submissions_ = 0;
PFN_xrGetDisplayRefreshRateFB get_display_refresh_rate_ = nullptr;
PFN_xrPerfSettingsSetPerformanceLevelEXT set_performance_level_ = nullptr;
#if defined(_WIN32)
using ConvertDisplayTime = XrResult (XRAPI_PTR*)(XrInstance, XrTime, LARGE_INTEGER*);
#else
+32
View File
@@ -566,6 +566,38 @@ OpenXREventStatus OpenXRRuntime::PollEvents() {
Log(OpenXRLogLevel::Warning, message.str());
break;
}
case XR_TYPE_EVENT_DATA_PERF_SETTINGS_EXT: {
// XR_EXT_performance_settings: the runtime reports when a domain's compositing,
// rendering or thermal state moves between normal, warning and impaired. Logged so a
// throttled headset explains a frame-rate drop in the session log.
const auto& perf_event =
*reinterpret_cast<const XrEventDataPerfSettingsEXT*>(&event);
const auto sub_domain = [](XrPerfSettingsSubDomainEXT value) {
switch (value) {
case XR_PERF_SETTINGS_SUB_DOMAIN_COMPOSITING_EXT: return "compositing";
case XR_PERF_SETTINGS_SUB_DOMAIN_RENDERING_EXT: return "rendering";
case XR_PERF_SETTINGS_SUB_DOMAIN_THERMAL_EXT: return "thermal";
default: return "unknown";
}
};
const auto notification = [](XrPerfSettingsNotificationLevelEXT value) {
switch (value) {
case XR_PERF_SETTINGS_NOTIF_LEVEL_NORMAL_EXT: return "normal";
case XR_PERF_SETTINGS_NOTIF_LEVEL_WARNING_EXT: return "warning";
case XR_PERF_SETTINGS_NOTIF_LEVEL_IMPAIRED_EXT: return "impaired";
default: return "unknown";
}
};
std::ostringstream message;
message << "OpenXR performance notification: "
<< (perf_event.domain == XR_PERF_SETTINGS_DOMAIN_CPU_EXT ? "CPU" : "GPU") << " "
<< sub_domain(perf_event.subDomain) << " " << notification(perf_event.fromLevel)
<< " -> " << notification(perf_event.toLevel);
Log(perf_event.toLevel == XR_PERF_SETTINGS_NOTIF_LEVEL_NORMAL_EXT ? OpenXRLogLevel::Info
: OpenXRLogLevel::Warning,
message.str());
break;
}
default:
break;
}