mirror of
https://github.com/mitch030504/Wiicompiled_VR_Frame.git
synced 2026-10-06 05:00:27 +02:00
Added performance level configuration for OpenXR runtime and enhance Aurora's frame worker
This commit is contained in:
1 parent
d296b3dcb3
commit
60c443654f
12 files changed
+292
-54
No files matched your search
@@ -49,6 +49,7 @@ first_person_head_right_meters = 0.0
|
||||
first_person_hide_driver = true
|
||||
first_person_hidden_model = 0
|
||||
first_person_rotation = "yaw"
|
||||
performance_level = "boost"
|
||||
```
|
||||
|
||||
To play this installation on the desktop instead, set `enabled = false`, close the game completely,
|
||||
@@ -127,6 +128,12 @@ desktop. `skip_copy_clears` independently suppresses the EFB reset performed aft
|
||||
default on and can be changed live from the F10 settings bar for diagnostics.
|
||||
`first_person` and the `first_person_*` values are the first-person camera described below. All
|
||||
four are live and are also exposed in the F10 settings bar.
|
||||
`performance_level` is the level asked of the runtime through `XR_EXT_performance_settings` for
|
||||
its CPU and GPU domains: `boost`, `sustained_high`, `sustained_low`, `power_savings`, or
|
||||
`default` to leave the runtime's own choice. Standalone headsets clock their cores by this
|
||||
request (see `docs/quest-port.md`); desktop runtimes rarely offer the extension, and the setting
|
||||
then does nothing. It is read at launch, and the session log records whether the runtime accepted
|
||||
it and any later performance notification (a thermal or rendering warning).
|
||||
|
||||
## Controllers
|
||||
|
||||
@@ -475,8 +482,10 @@ Stereo uniform calculations use cached CPU memory, followed by a single write in
|
||||
buffer. Reading or modifying matrices directly in D3D12 upload memory can be extremely slow,
|
||||
especially with many character draws; see Microsoft's [Map guidance](https://learn.microsoft.com/en-us/windows/win32/api/d3d12/nf-d3d12-id3d12resource-map).
|
||||
Retained interpolation reserves eye ranges at seal time and fills them once at the headset sample
|
||||
time. VR interpolation also releases the producer after sealing so eye encoding can overlap the
|
||||
next game frame, as it does with desktop interpolation.
|
||||
time. The frame worker always releases the producer after sealing, so eye encoding overlaps the
|
||||
next game frame whether or not interpolation is on. It used to publish that phase only after the
|
||||
encode unless interpolation was enabled, and the producer's first GX drain of every frame then
|
||||
waited for the previous frame's whole encode and submit (3 to 4.5 ms per frame on a Quest 3).
|
||||
|
||||
When VR interpolation is enabled at batch start, uniform recording also uses cached CPU
|
||||
memory. Matching and history capture read that buffer, then the used prefix is copied to
|
||||
|
||||
@@ -165,6 +165,16 @@ class SettingsPage(
|
||||
format = { "%.2f×".format(it) },
|
||||
write = { c, value -> c.setFloat("vr", "render_scale", value) },
|
||||
)
|
||||
choice(
|
||||
R.string.vr_performance_level, R.string.vr_performance_level_helper,
|
||||
listOf(
|
||||
R.string.vr_performance_level_boost, R.string.vr_performance_level_sustained_high,
|
||||
R.string.vr_performance_level_sustained_low, R.string.vr_performance_level_power_savings,
|
||||
R.string.vr_performance_level_default,
|
||||
),
|
||||
read = { stringIndex(it, "vr", "performance_level", PERFORMANCE_LEVELS) },
|
||||
write = { c, index -> c.setString("vr", "performance_level", PERFORMANCE_LEVELS[index]) },
|
||||
)
|
||||
choice(
|
||||
R.string.vr_interpolation, R.string.vr_interpolation_helper,
|
||||
listOf(activity.getString(R.string.vr_interpolation_off), activity.getString(R.string.vr_interpolation_auto), "72 FPS", "90 FPS", "120 FPS"),
|
||||
@@ -667,6 +677,8 @@ class SettingsPage(
|
||||
|
||||
private companion object {
|
||||
val ROTATIONS = listOf("yaw", "yaw_pitch", "full")
|
||||
// The runtime's default ("boost") first: an absent key reads as index 0.
|
||||
val PERFORMANCE_LEVELS = listOf("boost", "sustained_high", "sustained_low", "power_savings", "default")
|
||||
val CONTROLLER_MODES = listOf("wii_remote", "gamepad")
|
||||
val INTERPOLATION_FPS = listOf(0L, 1L, 72L, 90L, 120L)
|
||||
val RESOLUTIONS = listOf(1.0, 1.5, 2.0, 3.0, 4.0)
|
||||
|
||||
@@ -157,6 +157,13 @@
|
||||
<string name="section_vr_headset">Headset</string>
|
||||
<string name="vr_render_scale">Headset render scale</string>
|
||||
<string name="vr_render_scale_helper">Scales the eye resolution the headset recommends. Higher is sharper and costs GPU time; lower it if races stutter.</string>
|
||||
<string name="vr_performance_level">Headset performance level</string>
|
||||
<string name="vr_performance_level_helper">How hard the headset clocks its CPU and GPU for the game. Boost gives the most speed; lower it if the headset gets hot or the battery matters more.</string>
|
||||
<string name="vr_performance_level_boost">Boost</string>
|
||||
<string name="vr_performance_level_sustained_high">Sustained high</string>
|
||||
<string name="vr_performance_level_sustained_low">Sustained low</string>
|
||||
<string name="vr_performance_level_power_savings">Power savings</string>
|
||||
<string name="vr_performance_level_default">Headset default</string>
|
||||
<string name="vr_interpolation">VR frame interpolation</string>
|
||||
<string name="vr_interpolation_helper">Experimental. Renders extra frames between game frames at the headset\'s rate. Needs GPU headroom and adds one game frame of latency.</string>
|
||||
<string name="vr_interpolation_off">Off</string>
|
||||
|
||||
+63
-31
@@ -443,9 +443,28 @@ bool frame_worker_phase_reached(FrameWorkerPhase phase) noexcept {
|
||||
|
||||
bool wait_for_frame_worker_private_for(FrameWorkerPhase phase, std::chrono::microseconds timeout) noexcept;
|
||||
|
||||
// Where a frame's wall-clock time goes between the producer and the worker, summed in nanoseconds
|
||||
// and reported with the frame-rate log: the producer's waits for each worker phase, and the
|
||||
// worker cycle's four stretches: sealing, the wait for the producer's prepare permit, preparing
|
||||
// the next frame (which includes waiting for a free staging buffer), and the overlapped encode.
|
||||
std::atomic<uint64_t> g_producerWaitDoneNs{0};
|
||||
std::atomic<uint64_t> g_producerWaitSealedNs{0};
|
||||
std::atomic<uint64_t> g_workerSealNs{0};
|
||||
std::atomic<uint64_t> g_workerEncodeNs{0};
|
||||
std::atomic<uint64_t> g_workerPermitWaitNs{0};
|
||||
std::atomic<uint64_t> g_workerPrepareNs{0};
|
||||
|
||||
void wait_for_frame_worker_private(FrameWorkerPhase phase) noexcept {
|
||||
constexpr auto kWaitServiceInterval = std::chrono::milliseconds(1);
|
||||
if (frame_worker_phase_reached(phase)) {
|
||||
return;
|
||||
}
|
||||
const auto started = std::chrono::steady_clock::now();
|
||||
while (!wait_for_frame_worker_private_for(phase, kWaitServiceInterval)) {}
|
||||
const auto waited = static_cast<uint64_t>(
|
||||
std::chrono::duration_cast<std::chrono::nanoseconds>(std::chrono::steady_clock::now() - started).count());
|
||||
(phase == FrameWorkerPhase::Sealed ? g_producerWaitSealedNs : g_producerWaitDoneNs)
|
||||
.fetch_add(waited, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
bool wait_for_frame_worker_private_for(FrameWorkerPhase phase, std::chrono::microseconds timeout) noexcept {
|
||||
@@ -2169,8 +2188,21 @@ void record_frame_telemetry() {
|
||||
const auto now = std::chrono::steady_clock::now();
|
||||
const std::chrono::duration<double> elapsed = now - windowStart;
|
||||
if (elapsed.count() >= 5.0) {
|
||||
Log.info("Game frame rate {:.1f} FPS ({} frames in {:.2f} s)", windowFrames / elapsed.count(), windowFrames,
|
||||
elapsed.count());
|
||||
// Per-frame averages of where the wall-clock time went (see the counters' definition).
|
||||
const auto msPerFrame = [&](std::atomic<uint64_t>& counter) {
|
||||
return static_cast<double>(counter.exchange(0, std::memory_order_relaxed)) / 1e6 / std::max(windowFrames, 1u);
|
||||
};
|
||||
const double waitDone = msPerFrame(g_producerWaitDoneNs);
|
||||
const double waitSealed = msPerFrame(g_producerWaitSealedNs);
|
||||
const double seal = msPerFrame(g_workerSealNs);
|
||||
const double permitWait = msPerFrame(g_workerPermitWaitNs);
|
||||
const double prepare = msPerFrame(g_workerPrepareNs);
|
||||
const double encode = msPerFrame(g_workerEncodeNs);
|
||||
Log.info("Game frame rate {:.1f} FPS ({} frames in {:.2f} s); per frame the producer waited {:.2f} ms for "
|
||||
"DONE and {:.2f} ms for SEALED; the worker spent {:.2f} ms sealing, {:.2f} ms waiting for the "
|
||||
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding",
|
||||
windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal,
|
||||
permitWait, prepare, encode);
|
||||
windowStart = now;
|
||||
windowFrames = 0;
|
||||
}
|
||||
@@ -2187,19 +2219,17 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
ZoneScopedN("Frame worker cycle");
|
||||
webgpu::fail_if_device_lost();
|
||||
SealedFrameContext ctx;
|
||||
std::vector<PresentationJob> presentationJobs;
|
||||
bool overlapEncode = false;
|
||||
const auto elapsedNs = [](std::chrono::steady_clock::time_point since) {
|
||||
return static_cast<uint64_t>(
|
||||
std::chrono::duration_cast<std::chrono::nanoseconds>(std::chrono::steady_clock::now() - since).count());
|
||||
};
|
||||
auto stretchStarted = std::chrono::steady_clock::now();
|
||||
{
|
||||
std::lock_guard gpuLock(g_rendererGpuMutex);
|
||||
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
|
||||
overlapEncode = ctx.interpolationActive || ctx.retainStereo;
|
||||
if (!overlapEncode) {
|
||||
presentationJobs = encode_sealed_frame(sealedFrame, ctx);
|
||||
}
|
||||
}
|
||||
if (!overlapEncode) {
|
||||
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
|
||||
}
|
||||
g_workerSealNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
|
||||
stretchStarted = std::chrono::steady_clock::now();
|
||||
|
||||
{
|
||||
std::unique_lock lock(g_frameWorker.mutex);
|
||||
@@ -2209,39 +2239,41 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
}
|
||||
g_frameWorker.prepareAllowed = false;
|
||||
}
|
||||
g_workerPermitWaitNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
|
||||
stretchStarted = std::chrono::steady_clock::now();
|
||||
|
||||
// Preparing the next frame belongs to the SEALED phase: without a fresh pass 0 and mapped
|
||||
// staging buffers the producer's drain has nowhere to put its commands.
|
||||
bool imguiNewFrameOwed = false;
|
||||
const bool prepared = begin_frame_impl(
|
||||
false, overlapEncode ? ImGuiFramePolicy::Deferred : ImGuiFramePolicy::Immediate, &imguiNewFrameOwed);
|
||||
const bool prepared = begin_frame_impl(false, ImGuiFramePolicy::Deferred, &imguiNewFrameOwed);
|
||||
g_workerPrepareNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
|
||||
stretchStarted = std::chrono::steady_clock::now();
|
||||
|
||||
// SEALED before the encode, always. The encode reads only `ctx` and the sealed passes, so it
|
||||
// runs while the producer records the next frame, which takes the renderer mutex per drain.
|
||||
// It used to be published after the encode unless interpolation was on, and the producer's
|
||||
// first drain of each frame then waited for the whole encode and submit: 3 to 4.5 ms of every
|
||||
// frame on a Quest 3, the difference between a twelve-kart race start at 50 and at 60 fps.
|
||||
{
|
||||
std::lock_guard lock(g_frameWorker.mutex);
|
||||
g_frameWorker.framePrepared = prepared;
|
||||
g_frameWorker.sealed.store(true, std::memory_order_release);
|
||||
if (!overlapEncode) {
|
||||
g_frameWorker.ready.store(true, std::memory_order_release);
|
||||
}
|
||||
}
|
||||
g_frameWorker.cv.notify_all();
|
||||
|
||||
if (overlapEncode) {
|
||||
// Mutex-free: the producer drains and records the next frame in parallel, taking the renderer
|
||||
// mutex per drain, and this phase never takes it.
|
||||
presentationJobs = encode_sealed_frame(sealedFrame, ctx);
|
||||
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
|
||||
if (imguiNewFrameOwed) {
|
||||
// Safe only now: every slot has replayed this frame's ImGui draw lists.
|
||||
std::lock_guard gpuLock(g_rendererGpuMutex);
|
||||
imgui::new_frame(window::get_window_size());
|
||||
}
|
||||
{
|
||||
std::lock_guard lock(g_frameWorker.mutex);
|
||||
g_frameWorker.ready.store(true, std::memory_order_release);
|
||||
}
|
||||
g_frameWorker.cv.notify_all();
|
||||
std::vector<PresentationJob> presentationJobs = encode_sealed_frame(sealedFrame, ctx);
|
||||
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
|
||||
if (imguiNewFrameOwed) {
|
||||
// Safe only now: every slot has replayed this frame's ImGui draw lists.
|
||||
std::lock_guard gpuLock(g_rendererGpuMutex);
|
||||
imgui::new_frame(window::get_window_size());
|
||||
}
|
||||
g_workerEncodeNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
|
||||
{
|
||||
std::lock_guard lock(g_frameWorker.mutex);
|
||||
g_frameWorker.ready.store(true, std::memory_order_release);
|
||||
}
|
||||
g_frameWorker.cv.notify_all();
|
||||
|
||||
record_frame_telemetry();
|
||||
return true;
|
||||
|
||||
@@ -1833,16 +1833,44 @@ static void handle_draw_overrun(u8 cmd, u16 vtxCount, u32 vtxSize, u32 totalVtxB
|
||||
// Uploads a draw's GX vertices with the stride populate_pipeline_config gave the shader. When that stride is padded
|
||||
// (Android, see padded_upload_stride), each vertex is copied with zeroed trailing bytes; attribute offsets inside the
|
||||
// vertex are unchanged. Every padded upload is a multiple of 4 bytes, so consecutive draws stay contiguous for merging.
|
||||
// One padded element: srcStride bytes copied in word, half-word and byte steps, then the
|
||||
// padding zeroed. Inline arithmetic instead of a memcpy and memset call per element: a race
|
||||
// frame uploads tens of thousands of 7-byte vertices and 6-byte normals this way, and the two
|
||||
// library calls per element were most of the game thread's memcpy time on the Quest.
|
||||
static inline void copy_padded_element(u8* dst, const u8* src, u32 srcStride, u32 dstStride) noexcept {
|
||||
u32 n = srcStride;
|
||||
while (n >= 4) {
|
||||
u32 word;
|
||||
std::memcpy(&word, src, 4);
|
||||
std::memcpy(dst, &word, 4);
|
||||
src += 4;
|
||||
dst += 4;
|
||||
n -= 4;
|
||||
}
|
||||
if (n >= 2) {
|
||||
u16 half;
|
||||
std::memcpy(&half, src, 2);
|
||||
std::memcpy(dst, &half, 2);
|
||||
src += 2;
|
||||
dst += 2;
|
||||
n -= 2;
|
||||
}
|
||||
if (n != 0) {
|
||||
*dst++ = *src;
|
||||
}
|
||||
for (u32 pad = srcStride; pad < dstStride; ++pad) {
|
||||
*dst++ = 0;
|
||||
}
|
||||
}
|
||||
|
||||
static gfx::Range push_draw_vertices(const u8* vertices, u32 vtxCount, u32 vtxSize) {
|
||||
const u32 uploadStride = padded_upload_stride(vtxSize);
|
||||
if (uploadStride == vtxSize)
|
||||
LIKELY { return gfx::push_verts(vertices, static_cast<size_t>(vtxCount) * vtxSize); }
|
||||
auto [buffer, range] = gfx::map_verts(static_cast<size_t>(vtxCount) * uploadStride);
|
||||
u8* dst = buffer.data();
|
||||
const u32 padding = uploadStride - vtxSize;
|
||||
for (u32 i = 0; i < vtxCount; ++i) {
|
||||
std::memcpy(dst, vertices + static_cast<size_t>(i) * vtxSize, vtxSize);
|
||||
std::memset(dst + vtxSize, 0, padding);
|
||||
copy_padded_element(dst, vertices + static_cast<size_t>(i) * vtxSize, vtxSize, uploadStride);
|
||||
dst += uploadStride;
|
||||
}
|
||||
return range;
|
||||
@@ -1857,10 +1885,15 @@ static gfx::Range push_vertex_array(const AttrArray& array, u32 uploadStride) {
|
||||
const size_t count = (static_cast<size_t>(array.size) + array.stride - 1) / array.stride;
|
||||
auto [buffer, range] = gfx::map_storage(count * uploadStride);
|
||||
u8* dst = buffer.data();
|
||||
std::memset(dst, 0, count * uploadStride);
|
||||
for (size_t i = 0; i < count; ++i) {
|
||||
const size_t start = i * array.stride;
|
||||
std::memcpy(dst + i * uploadStride, data + start, std::min<size_t>(array.stride, array.size - start));
|
||||
// Every whole element inline; only a trailing partial element takes the library calls.
|
||||
const size_t whole = static_cast<size_t>(array.size) / array.stride;
|
||||
for (size_t i = 0; i < whole; ++i) {
|
||||
copy_padded_element(dst + i * uploadStride, data + i * array.stride, array.stride, uploadStride);
|
||||
}
|
||||
if (whole < count) {
|
||||
const size_t start = whole * array.stride;
|
||||
std::memset(dst + whole * uploadStride, 0, uploadStride);
|
||||
std::memcpy(dst + whole * uploadStride, data + start, array.size - start);
|
||||
}
|
||||
return range;
|
||||
}
|
||||
|
||||
+28
-1
@@ -582,7 +582,7 @@ the app:
|
||||
| `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update |
|
||||
| `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds |
|
||||
| `debug.wiicompiled.inject <n>:<button>` | Presses `a`, `b`, `x`, `y`, `start`, `up`, `down`, `left` or `right` for 12 XR frames each time `<n>` changes. As a Wii Remote, `x`/`y`/`start` are 1/2/+, the directions push the Nunchuk stick, and `home`, `c` and `z` also exist. `panel` clicks both thumbsticks, opening or closing the settings panel (see `OPENXR.md`) |
|
||||
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s. The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%` and app GPU time (`App=`) |
|
||||
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
|
||||
|
||||
The injector makes headset tests possible with nobody wearing the headset.
|
||||
Keep the display awake, drive the menus, then take a compositor screenshot:
|
||||
@@ -622,6 +622,33 @@ follows the game rate instead of holding 72 Hz, so the pacing thread is not
|
||||
repeating the last layer as it does on desktop. Both need profiling on the
|
||||
XR2 Gen 2.
|
||||
|
||||
Profiled 2026-09-19 on a twelve-kart 50cc Grand Prix start (Luigi Circuit,
|
||||
`render_scale` 0.8, intro skipped, driven unattended by the button injector:
|
||||
ten `a` presses 5 s apart from the title screen reach the race, one more skips
|
||||
the course intro). The game thread is the limit, not the GPU: it is one
|
||||
libco-hosted thread (about 60% translated game code, 20% GX HLE, 12% Aurora's
|
||||
FIFO decode), and Aurora's frame worker, which encodes and submits the Dawn
|
||||
work, runs at about half a core with most of its own time inside the Adreno
|
||||
driver's ioctls. Trimming the GX HLE (a 4 KiB write-tracking granule, one
|
||||
pointer probe for the texture-object shadow, inline padded vertex copies, a
|
||||
throttled clock poll) changed nothing measurable against the same automated
|
||||
start, and asking for `XR_EXT_performance_settings` BOOST is accepted but the
|
||||
runtime keeps its own dynamic clocks (CPU level 4 at 1.9 to 2.2 GHz, GPU level
|
||||
3 at 490 to 640 MHz). What did matter was a scheduler trace of the game
|
||||
thread: it slept 3 to 4.5 ms of every frame, in 1 ms slices, on Aurora's
|
||||
SEALED phase. Without interpolation the worker published SEALED only after the
|
||||
whole encode and submit, so the producer's first GX drain of each frame waited
|
||||
for the previous frame's encode (5 ms in menus, 9 to 11 ms in a race). The
|
||||
worker now always releases the producer right after sealing; the same start
|
||||
went from 47 to 53 fps to 55 to 58 fps, the game thread from 82% to 98% busy,
|
||||
and menus lost the same 4.5 ms of idle wait per frame. With `fpslog` on, the
|
||||
`Game frame rate` line now carries that breakdown (producer waits for DONE and
|
||||
SEALED; the worker's seal, permit wait, prepare and encode) so the next
|
||||
regression of this kind shows up in the session log. The remaining gap to 60
|
||||
at the start is about 1 ms of game-thread CPU per frame, with the GPU at 85 to
|
||||
89%, so the next steps are on both sides: the guest-code share (translator
|
||||
output quality) and the eye replay's GPU cost.
|
||||
|
||||
Verified on device since: the menus on the virtual screen, controller input
|
||||
(the user has driven races), and an immersive Grand Prix start with all 12
|
||||
racers rendering correctly. Not yet verified: stereo comfort and scale,
|
||||
|
||||
@@ -39,10 +39,12 @@ inline uint32_t CanonicalizeGxMainRamAddress(uint32_t addr) noexcept {
|
||||
|
||||
namespace GxGuestWrite {
|
||||
|
||||
// Granularity matches the GX resource caches' occupancy maps: 64 KiB, so a
|
||||
// cacheable display list or texture spans only a handful of counters. A false
|
||||
// bump only costs one re-digest, which is exactly the untracked behaviour.
|
||||
inline constexpr uint32_t kGranuleShift = 16; // 64 KiB per granule
|
||||
// 4 KiB per granule, so a cacheable display list or texture spans a few counters and a false
|
||||
// bump only costs one re-digest, which is exactly the untracked behaviour. It used to be 64 KiB:
|
||||
// then per-frame guest writes landing beside a static list or texture (a race start streams
|
||||
// data next to them) bumped the same granule, and every affected list was re-digested and every
|
||||
// affected texture re-hashed each frame. The table is 320 KiB.
|
||||
inline constexpr uint32_t kGranuleShift = 12;
|
||||
inline constexpr uint64_t kTrackedSpan = static_cast<uint64_t>(Memory::kMem2PhysicalEnd);
|
||||
inline constexpr size_t kGranuleCount = static_cast<size_t>(kTrackedSpan >> kGranuleShift);
|
||||
|
||||
|
||||
@@ -68,6 +68,7 @@ struct RuntimeUserConfig {
|
||||
std::optional<bool> vrFirstPersonHideDriver;
|
||||
std::optional<int32_t> vrFirstPersonHiddenModel;
|
||||
std::optional<std::string> vrFirstPersonRotation;
|
||||
std::optional<std::string> vrPerformanceLevel;
|
||||
std::optional<std::string> vrRecenterKey;
|
||||
std::optional<float> vrLeanBackDegrees;
|
||||
// F10 > Diagnostics: OpenXR pacing and presentation logging in console.log.
|
||||
@@ -179,6 +180,16 @@ inline constexpr const char* kVrFirstPersonRotationDefault = "yaw";
|
||||
inline bool IsSupportedVrFirstPersonRotation(std::string_view value) {
|
||||
return value == "yaw" || value == "yaw_pitch" || value == "full";
|
||||
}
|
||||
// The performance level asked of the OpenXR runtime (XR_EXT_performance_settings) for its CPU and
|
||||
// GPU domains. Standalone headsets clock their cores by this request: a Quest 3 ran the game
|
||||
// thread at 1.92 GHz with the runtime's own choice while its fast cores reach 2.36 GHz. "default"
|
||||
// leaves the runtime's choice; desktop runtimes without the extension ignore the setting.
|
||||
inline constexpr const char* kVrPerformanceLevelDefault = "boost";
|
||||
|
||||
inline bool IsSupportedVrPerformanceLevel(std::string_view value) {
|
||||
return value == "default" || value == "power_savings" || value == "sustained_low" ||
|
||||
value == "sustained_high" || value == "boost";
|
||||
}
|
||||
// What the desktop window shows while the headset is running: "normal" leaves
|
||||
// the ordinary desktop view alone, "both", "left" and "right" mirror the
|
||||
// headset's eyes, and "none" blacks the window out. Matches
|
||||
@@ -455,6 +466,11 @@ inline void EnsureConfigFile() {
|
||||
"# horizon, \"yaw_pitch\" adds the kart's climb but no roll, and\n"
|
||||
"# \"full\" takes the kart's whole orientation so the view banks.\n"
|
||||
"first_person_rotation = \"yaw\"\n\n"
|
||||
"# Performance level asked of the headset's runtime for its CPU and\n"
|
||||
"# GPU: \"boost\", \"sustained_high\", \"sustained_low\", \"power_savings\",\n"
|
||||
"# or \"default\" to leave the runtime's own choice. Standalone headsets\n"
|
||||
"# clock their cores by this request; desktop runtimes ignore it.\n"
|
||||
"performance_level = \"boost\"\n\n"
|
||||
"# Keyboard shortcut that recenters the VR view, naming the key the\n"
|
||||
"# way SDL does (F9, Home, Keypad 5, ...). It moves the race view to\n"
|
||||
"# where you are sitting now and brings the menu screen back upright in\n"
|
||||
@@ -675,6 +691,10 @@ inline RuntimeUserConfig ParseConfigDocument(const toml::value& document) {
|
||||
value && IsSupportedVrFirstPersonRotation(*value)) {
|
||||
config.vrFirstPersonRotation = *value;
|
||||
}
|
||||
if (auto value = FindConfigValue<std::string>(document, "vr", "performance_level");
|
||||
value && IsSupportedVrPerformanceLevel(*value)) {
|
||||
config.vrPerformanceLevel = *value;
|
||||
}
|
||||
if (auto value = FindConfigValue<std::string>(document, "vr", "mirror_view");
|
||||
value && IsSupportedVrMirrorView(*value)) {
|
||||
config.vrMirrorView = *value;
|
||||
@@ -1013,6 +1033,14 @@ inline bool SetVrFirstPersonRotation(std::string value) {
|
||||
return WriteSetting("vr", "first_person_rotation", FormatString(value));
|
||||
}
|
||||
|
||||
inline bool SetVrPerformanceLevel(std::string value) {
|
||||
if (!IsSupportedVrPerformanceLevel(value)) {
|
||||
return false;
|
||||
}
|
||||
Mutable().vrPerformanceLevel = value;
|
||||
return WriteSetting("vr", "performance_level", FormatString(value));
|
||||
}
|
||||
|
||||
inline bool SetVrFirstPersonHiddenModel(int32_t value) {
|
||||
value = std::clamp(value, -1, 31);
|
||||
Mutable().vrFirstPersonHiddenModel = value;
|
||||
@@ -1371,6 +1399,11 @@ inline std::string VrFirstPersonRotation(std::string fallback = kVrFirstPersonRo
|
||||
return value && IsSupportedVrFirstPersonRotation(*value) ? *value : std::move(fallback);
|
||||
}
|
||||
|
||||
inline std::string VrPerformanceLevel(std::string fallback = kVrPerformanceLevelDefault) {
|
||||
const auto& value = Get().vrPerformanceLevel;
|
||||
return value && IsSupportedVrPerformanceLevel(*value) ? *value : std::move(fallback);
|
||||
}
|
||||
|
||||
inline int32_t VrFirstPersonHiddenModel(int32_t fallback = kVrFirstPersonHiddenModelDefault) {
|
||||
return std::clamp(Get().vrFirstPersonHiddenModel.value_or(fallback), -1, 31);
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
#include "runtime_log.h"
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
|
||||
// --- Texture and TLUT Objects ---
|
||||
@@ -447,15 +448,23 @@ TexObjMeta MergeGuestTexObjMeta(const TexObjMeta& cached, const TexObjMeta& gues
|
||||
|
||||
// Reads the raw 32 guest bytes of a GXTexObj struct. Returns false (and the
|
||||
// caller must treat the shadow as absent) when the address is unreadable.
|
||||
// The shadow is only ever compared for equality (ShadowEquals), so it keeps
|
||||
// the bytes in guest order: one page-table probe and one copy per lookup
|
||||
// instead of four checked, byte-swapped 64-bit reads. Every draw's texture
|
||||
// binds go through this on the game thread, so the per-call cost matters.
|
||||
bool ReadTexObjShadow(uint32_t addr, uint64_t (&out)[4]) noexcept {
|
||||
try {
|
||||
out[0] = Memory::Read64(addr + 0x00);
|
||||
out[1] = Memory::Read64(addr + 0x08);
|
||||
out[2] = Memory::Read64(addr + 0x10);
|
||||
out[3] = Memory::Read64(addr + 0x18);
|
||||
} catch (...) {
|
||||
return false;
|
||||
const uint8_t* bytes = MemoryInline::GetPointerFast(addr, sizeof(out));
|
||||
if (bytes == nullptr) {
|
||||
try {
|
||||
bytes = Memory::GetPointer(addr, sizeof(out));
|
||||
} catch (...) {
|
||||
return false;
|
||||
}
|
||||
if (bytes == nullptr) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
std::memcpy(out, bytes, sizeof(out));
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -244,6 +244,16 @@ void ServiceDeferredTimingDuringGxWork() {
|
||||
return;
|
||||
}
|
||||
|
||||
// GX__Begin runs thousands of times a frame, and reading the clock on each one was 2% of the
|
||||
// game thread on the Quest. The poll has a 1 ms cadence, so sampling the clock on every
|
||||
// sixteenth call keeps it within a few microseconds of that. A plain static: GX runs on the
|
||||
// one guest-facing thread, and a thread_local here would cost a resolver call per access.
|
||||
static uint32_t s_pollCountdown = 0;
|
||||
if (s_pollCountdown != 0) {
|
||||
--s_pollCountdown;
|
||||
return;
|
||||
}
|
||||
s_pollCountdown = 15;
|
||||
const auto now = std::chrono::steady_clock::now();
|
||||
if (now < g_nextDeferredTimingPoll) {
|
||||
return;
|
||||
|
||||
@@ -369,7 +369,7 @@ public:
|
||||
#if defined(_WIN32)
|
||||
config.required_extensions = {"XR_KHR_D3D12_enable"};
|
||||
config.optional_extensions = {"XR_KHR_win32_convert_performance_counter_time",
|
||||
"XR_FB_display_refresh_rate"};
|
||||
"XR_FB_display_refresh_rate", "XR_EXT_performance_settings"};
|
||||
#else
|
||||
// Either Vulkan binding extension is acceptable; the backend picks
|
||||
// whichever the runtime enabled, preferring enable2.
|
||||
@@ -377,7 +377,7 @@ public:
|
||||
config.optional_extensions = {"XR_KHR_vulkan_enable2", "XR_KHR_vulkan_enable",
|
||||
"XR_KHR_convert_timespec_time",
|
||||
"XR_KHR_android_thread_settings",
|
||||
"XR_FB_display_refresh_rate"};
|
||||
"XR_FB_display_refresh_rate", "XR_EXT_performance_settings"};
|
||||
config.instance_create_next = OpenXRAndroidInstanceCreateNext();
|
||||
#endif
|
||||
if (!runtime_->Initialize(config)) {
|
||||
@@ -401,6 +401,9 @@ public:
|
||||
if (has_extension("XR_FB_display_refresh_rate")) {
|
||||
runtime_->LoadFunction("xrGetDisplayRefreshRateFB", &get_display_refresh_rate_);
|
||||
}
|
||||
if (has_extension("XR_EXT_performance_settings")) {
|
||||
runtime_->LoadFunction("xrPerfSettingsSetPerformanceLevelEXT", &set_performance_level_);
|
||||
}
|
||||
interpolation_available_.store(convert_display_time_ != nullptr, std::memory_order_release);
|
||||
if (!backend_->QueryGraphicsRequirements(*runtime_)) {
|
||||
SetError(backend_->LastError());
|
||||
@@ -509,6 +512,7 @@ public:
|
||||
prepared_ = false;
|
||||
convert_display_time_ = nullptr;
|
||||
get_display_refresh_rate_ = nullptr;
|
||||
set_performance_level_ = nullptr;
|
||||
headset_hz_.store(0, std::memory_order_relaxed);
|
||||
rendered_fps_.store(0, std::memory_order_relaxed);
|
||||
interpolation_available_.store(false, std::memory_order_release);
|
||||
@@ -635,6 +639,32 @@ private:
|
||||
}
|
||||
#endif
|
||||
|
||||
// Asks the runtime for the configured performance level in both domains. Standalone
|
||||
// headsets clock their cores by this: a Quest 3 held the game thread at CPU level 4
|
||||
// (2.2 GHz of a possible 2.36) and the GPU at level 3 with the runtime's own choice. A
|
||||
// refusal is logged and changes nothing; desktop runtimes rarely offer the extension.
|
||||
void ApplyPerformanceLevel() {
|
||||
if (runtime_ == nullptr || set_performance_level_ == nullptr || !runtime_->HasSession()) {
|
||||
return;
|
||||
}
|
||||
const std::string requested = RuntimeConfigFile::VrPerformanceLevel();
|
||||
XrPerfSettingsLevelEXT level = XR_PERF_SETTINGS_LEVEL_SUSTAINED_HIGH_EXT;
|
||||
if (requested == "default") {
|
||||
return;
|
||||
} else if (requested == "power_savings") {
|
||||
level = XR_PERF_SETTINGS_LEVEL_POWER_SAVINGS_EXT;
|
||||
} else if (requested == "sustained_low") {
|
||||
level = XR_PERF_SETTINGS_LEVEL_SUSTAINED_LOW_EXT;
|
||||
} else if (requested == "boost") {
|
||||
level = XR_PERF_SETTINGS_LEVEL_BOOST_EXT;
|
||||
}
|
||||
const XrResult cpu = set_performance_level_(runtime_->Session(), XR_PERF_SETTINGS_DOMAIN_CPU_EXT, level);
|
||||
const XrResult gpu = set_performance_level_(runtime_->Session(), XR_PERF_SETTINGS_DOMAIN_GPU_EXT, level);
|
||||
RT_LOG(RT_TAG_RUNTIME) << "OpenXR: performance level \"" << requested << "\" CPU "
|
||||
<< (XR_SUCCEEDED(cpu) ? "set" : "refused") << " (" << cpu << "), GPU "
|
||||
<< (XR_SUCCEEDED(gpu) ? "set" : "refused") << " (" << gpu << ")" << std::endl;
|
||||
}
|
||||
|
||||
static bool ProvideStereoFrame(uint32_t, AuroraStereoFrame* output, void* userdata) {
|
||||
auto* self = static_cast<OpenXRIntegration*>(userdata);
|
||||
if (self == nullptr || output == nullptr) {
|
||||
@@ -672,6 +702,7 @@ private:
|
||||
worker_registered = RegisterAuroraFrameWorkerThread();
|
||||
}
|
||||
#endif
|
||||
ApplyPerformanceLevel();
|
||||
bool fatal = false;
|
||||
uint32_t consecutive_skips = 0;
|
||||
bool store_gate_set = false;
|
||||
@@ -1295,6 +1326,7 @@ private:
|
||||
std::chrono::steady_clock::time_point timing_start_ = std::chrono::steady_clock::now();
|
||||
uint32_t timing_submissions_ = 0;
|
||||
PFN_xrGetDisplayRefreshRateFB get_display_refresh_rate_ = nullptr;
|
||||
PFN_xrPerfSettingsSetPerformanceLevelEXT set_performance_level_ = nullptr;
|
||||
#if defined(_WIN32)
|
||||
using ConvertDisplayTime = XrResult (XRAPI_PTR*)(XrInstance, XrTime, LARGE_INTEGER*);
|
||||
#else
|
||||
|
||||
@@ -566,6 +566,38 @@ OpenXREventStatus OpenXRRuntime::PollEvents() {
|
||||
Log(OpenXRLogLevel::Warning, message.str());
|
||||
break;
|
||||
}
|
||||
case XR_TYPE_EVENT_DATA_PERF_SETTINGS_EXT: {
|
||||
// XR_EXT_performance_settings: the runtime reports when a domain's compositing,
|
||||
// rendering or thermal state moves between normal, warning and impaired. Logged so a
|
||||
// throttled headset explains a frame-rate drop in the session log.
|
||||
const auto& perf_event =
|
||||
*reinterpret_cast<const XrEventDataPerfSettingsEXT*>(&event);
|
||||
const auto sub_domain = [](XrPerfSettingsSubDomainEXT value) {
|
||||
switch (value) {
|
||||
case XR_PERF_SETTINGS_SUB_DOMAIN_COMPOSITING_EXT: return "compositing";
|
||||
case XR_PERF_SETTINGS_SUB_DOMAIN_RENDERING_EXT: return "rendering";
|
||||
case XR_PERF_SETTINGS_SUB_DOMAIN_THERMAL_EXT: return "thermal";
|
||||
default: return "unknown";
|
||||
}
|
||||
};
|
||||
const auto notification = [](XrPerfSettingsNotificationLevelEXT value) {
|
||||
switch (value) {
|
||||
case XR_PERF_SETTINGS_NOTIF_LEVEL_NORMAL_EXT: return "normal";
|
||||
case XR_PERF_SETTINGS_NOTIF_LEVEL_WARNING_EXT: return "warning";
|
||||
case XR_PERF_SETTINGS_NOTIF_LEVEL_IMPAIRED_EXT: return "impaired";
|
||||
default: return "unknown";
|
||||
}
|
||||
};
|
||||
std::ostringstream message;
|
||||
message << "OpenXR performance notification: "
|
||||
<< (perf_event.domain == XR_PERF_SETTINGS_DOMAIN_CPU_EXT ? "CPU" : "GPU") << " "
|
||||
<< sub_domain(perf_event.subDomain) << " " << notification(perf_event.fromLevel)
|
||||
<< " -> " << notification(perf_event.toLevel);
|
||||
Log(perf_event.toLevel == XR_PERF_SETTINGS_NOTIF_LEVEL_NORMAL_EXT ? OpenXRLogLevel::Info
|
||||
: OpenXRLogLevel::Warning,
|
||||
message.str());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user