From 47b59c7294662be64797d72b36196b398d001451 Mon Sep 17 00:00:00 2001 From: iChris4 Date: Sat, 19 Sep 2026 19:23:57 +0200 Subject: [PATCH] Implement per-pass GPU timing for performance tracking and optimization & fixed mono in Immersive --- aurora-main/lib/aurora.cpp | 32 +++- aurora-main/lib/gfx/common.cpp | 231 ++++++++++++++++++++++- aurora-main/lib/gfx/common.hpp | 43 ++++- aurora-main/lib/gfx/depth_peek.cpp | 2 + aurora-main/lib/gfx/tex_copy_conv.cpp | 2 + aurora-main/lib/gfx/tex_palette_conv.cpp | 2 + aurora-main/lib/stereo_overlay.cpp | 2 + aurora-main/lib/webgpu/gpu.cpp | 12 ++ aurora-main/lib/webgpu/gpu.hpp | 2 + docs/quest-port.md | 21 ++- 10 files changed, 344 insertions(+), 5 deletions(-) diff --git a/aurora-main/lib/aurora.cpp b/aurora-main/lib/aurora.cpp index b2f0a2b..9744665 100644 --- a/aurora-main/lib/aurora.cpp +++ b/aurora-main/lib/aurora.cpp @@ -756,6 +756,7 @@ void encode_virtual_screen_eye(wgpu::CommandEncoder& encoder, const webgpu::Pres .label = eyeIndex == 0 ? "Virtual screen left eye" : "Virtual screen right eye", .colorAttachmentCount = attachments.size(), .colorAttachments = attachments.data(), + .timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::VirtualScreen), }; { const auto pass = encoder.BeginRenderPass(&descriptor); @@ -1274,6 +1275,7 @@ bool present_presentation_job(const PresentationJob& job) { .label = "Presentation copy pass", .colorAttachmentCount = attachments.size(), .colorAttachments = attachments.data(), + .timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Present), }; const auto pass = encoder.BeginRenderPass(&renderPassDescriptor); pass.SetPipeline(webgpu::g_CopyPipeline); @@ -1576,6 +1578,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web .label = "Interpolation snapshot pass", .colorAttachmentCount = attachments.size(), .colorAttachments = attachments.data(), + .timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot), }; const auto pass = encoder.BeginRenderPass(&renderPassDescriptor); const auto imageWidth = static_cast(image.texture.size.width); @@ -1627,6 +1630,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web .label = "Snapshot ImGui pass", .colorAttachmentCount = attachments.size(), .colorAttachments = attachments.data(), + .timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot), }; const auto pass = encoder.BeginRenderPass(&renderPassDescriptor); pass.SetViewport(0.f, 0.f, static_cast(image.texture.size.width), @@ -1829,6 +1833,9 @@ void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept { void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag, const StereoSceneAnchor& sceneAnchor) { ZoneScopedN("Seal frame"); + // Every pass this cycle encodes, from the seal's probe blits to the final eye, is timed under + // one frame; encode_sealed_frame resolves it on its last submission. + gfx::gpu_timing_begin_frame(); const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{ .label = "Redraw encoder", }; @@ -1988,7 +1995,18 @@ std::vector encode_sealed_frame(gfx::SealedFrame& sealedFrame, // A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed // stream would mutate an already-rendered EFB. Render once, then duplicate into the slots. - gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo); + // + // On a headset an immersive frame's native render is never presented: the eyes replay the draws + // themselves and only sample the EFB copies it resolves. So it stops after the last pass that + // produces one of those copies (never the display copy), which on a Quest 3 was 4 to 6 ms of a + // 12 ms GPU frame spent on a 1280x720 image nobody saw. A pending CPU readback or a frame + // capture still gets the whole image. + int32_t nativeRenderLastPass = INT32_MAX; + if (headsetOnly && immersiveReplay && !gfx::efb_ram::has_pending() && + g_captureFrame.load(std::memory_order_acquire) == UINT32_MAX) { + nativeRenderLastPass = gfx::last_pass_feeding_replay(sealedFrame); + } + gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo, nativeRenderLastPass); // The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder; // completion is harvested in gfx::after_submit, never waited on here. gfx::efb_ram::encode_async_downloads(encoder); @@ -2070,7 +2088,9 @@ std::vector encode_sealed_frame(gfx::SealedFrame& sealedFrame, .presentAt = slotPresentDeadline(ctx.interpolatedFrameCount), .interpolated = false, }); + gfx::gpu_timing_end_frame(encoder); submitEncodedSlot(encoder, pendingStereoSink ? &*pendingStereoSink : nullptr); + gfx::gpu_timing_after_submit(); // A group that finished encoding past its anchor slides forward by whole display periods, never // per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period. @@ -2180,7 +2200,12 @@ void record_frame_telemetry() { { // `adb shell setprop debug.wiicompiled.fpslog 1` before launch logs the game's rendered frame rate every five // seconds. The headset compositor's own log (logcat tag VrApi) repeats frames, so it cannot show this. - static const bool fpsLog = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1; + static const bool fpsLog = [] { + const bool on = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1; + // The same switch turns on the per-pass GPU timestamps reported below the frame-rate line. + gfx::gpu_timing_set_enabled(on); + return on; + }(); if (fpsLog) { static auto windowStart = std::chrono::steady_clock::now(); static uint32_t windowFrames = 0; @@ -2203,6 +2228,9 @@ void record_frame_telemetry() { "prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding", windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal, permitWait, prepare, encode); + if (const std::string gpuTiming = gfx::gpu_timing_report(); !gpuTiming.empty()) { + Log.info("{}", gpuTiming); + } windowStart = now; windowFrames = 0; } diff --git a/aurora-main/lib/gfx/common.cpp b/aurora-main/lib/gfx/common.cpp index 1d9e499..951cbba 100644 --- a/aurora-main/lib/gfx/common.cpp +++ b/aurora-main/lib/gfx/common.cpp @@ -1724,6 +1724,9 @@ struct RenderInvocation { uint32_t localPlayerCount = 1; // Inclusive index of the last pass to replay; -1 replays every pass. int32_t replayLastPass = -1; + // Inclusive index of the last pass that does render work; texture bakes still run for the + // passes after it. See last_pass_feeding_replay. + int32_t renderLastPass = INT32_MAX; bool finalize = true; bool replayOnlyEfb = false; bool skipCopyClears = false; @@ -1755,6 +1758,11 @@ static void render_impl(std::vector& renderPasses, wgpu::CommandEnco tex_palette_conv::run(cmd, conv); } } + if (static_cast(i) > invocation.renderLastPass) { + // Nothing after the last replay-feeding resolve is shown or sampled on a headset; the + // bakes above are all these passes owe the eye replays. + continue; + } const bool hasRenderWork = passInfo.clearColor || passInfo.clearDepth || !passInfo.commands.empty(); if (i == renderPasses.size() - 1) { ASSERT(!passInfo.resolveTarget, "Final render pass must not have resolve target"); @@ -1793,11 +1801,16 @@ static void render_impl(std::vector& renderPasses, wgpu::CommandEnco .depthStoreOp = wgpu::StoreOp::Store, .depthClearValue = passInfo.clearDepthValue, }; + const GpuTimingCategory timingCategory = invocation.stereoEye == 0 ? GpuTimingCategory::EyeLeft + : invocation.stereoEye == 1 ? GpuTimingCategory::EyeRight + : invocation.interpolatedFrame >= 0 ? GpuTimingCategory::Interpolated + : GpuTimingCategory::Mono; const wgpu::RenderPassDescriptor renderPassDescriptor{ .label = render_pass_label(i), .colorAttachmentCount = attachments.size(), .colorAttachments = attachments.data(), .depthStencilAttachment = &depthStencilAttachment, + .timestampWrites = gpu_timing_pass(timingCategory), }; auto pass = cmd.BeginRenderPass(&renderPassDescriptor); @@ -1908,15 +1921,28 @@ void seal_frame(SealedFrame& out) noexcept { g_currentRenderPass = UINT32_MAX; } -void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) { +void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize, + int32_t nativeRenderLastPass) { render_impl(frame.data().passes, cmd, RenderInvocation{ .interpolatedFrame = interpolatedFrame, + .renderLastPass = nativeRenderLastPass, .finalize = finalize, .encodeTextureBakes = interpolatedFrame < 0, }); } +int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept { + const auto& passes = frame.data().passes; + int32_t last = -1; + for (size_t i = 0; i < passes.size(); ++i) { + if (passes[i].resolveTarget && !passes[i].displayCopyResolve) { + last = static_cast(i); + } + } + return last; +} + bool has_late_stereo_replay(const SealedFrame& frame) noexcept { const auto& data = frame.data().stereo; return data.generation != 0 && data.generation == g_replayBufferGeneration.load(std::memory_order_acquire) && @@ -2028,6 +2054,209 @@ void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) } } +// --- Per-pass GPU timing (see common.hpp) ------------------------------------------------------- +namespace { +constexpr uint32_t kGpuTimingSlots = 4; +constexpr uint32_t kGpuTimingPairs = 62; +constexpr uint32_t kGpuTimingQueries = 2 * kGpuTimingPairs; + +struct GpuTimingSlot { + wgpu::QuerySet querySet; + wgpu::Buffer resolve; + wgpu::Buffer readback; + std::array writes{}; + std::array categories{}; + uint32_t pairs = 0; + bool open = false; // between the frame's begin and end + bool reading = false; // readback in flight or mapped + bool mapped = false; // the callback ran; the encoding thread unmaps on reuse +}; + +std::atomic g_gpuTimingEnabled{false}; +std::array g_gpuTimingSlots; +uint32_t g_gpuTimingNextSlot = 0; +int32_t g_gpuTimingCurrent = -1; +bool g_gpuTimingReady = false; +// Guards the totals below and every slot's reading/mapped flags: the map callback may run on +// whichever thread processes Dawn's events. +std::mutex g_gpuTimingMutex; +std::array(GpuTimingCategory::Count)> g_gpuTimingTotalsNs{}; +uint64_t g_gpuTimingSpanNs = 0; +uint32_t g_gpuTimingFrames = 0; +uint32_t g_gpuTimingSkipped = 0; + +bool gpu_timing_create_slots() { + if (g_gpuTimingReady) { + return true; + } + if (!webgpu::g_timestampQueriesSupported || !webgpu::g_device) { + return false; + } + for (auto& slot : g_gpuTimingSlots) { + const wgpu::QuerySetDescriptor querySetDescriptor{ + .label = "GPU timing queries", + .type = wgpu::QueryType::Timestamp, + .count = kGpuTimingQueries, + }; + slot.querySet = webgpu::g_device.CreateQuerySet(&querySetDescriptor); + const wgpu::BufferDescriptor resolveDescriptor{ + .label = "GPU timing resolve", + .usage = wgpu::BufferUsage::QueryResolve | wgpu::BufferUsage::CopySrc, + .size = kGpuTimingQueries * sizeof(uint64_t), + }; + slot.resolve = webgpu::g_device.CreateBuffer(&resolveDescriptor); + const wgpu::BufferDescriptor readbackDescriptor{ + .label = "GPU timing readback", + .usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst, + .size = kGpuTimingQueries * sizeof(uint64_t), + }; + slot.readback = webgpu::g_device.CreateBuffer(&readbackDescriptor); + } + g_gpuTimingReady = true; + return true; +} +} // namespace + +void gpu_timing_set_enabled(bool enabled) noexcept { g_gpuTimingEnabled.store(enabled, std::memory_order_relaxed); } +bool gpu_timing_enabled() noexcept { return g_gpuTimingEnabled.load(std::memory_order_relaxed); } + +void gpu_timing_begin_frame() noexcept { + g_gpuTimingCurrent = -1; + if (!gpu_timing_enabled() || !gpu_timing_create_slots()) { + return; + } + const uint32_t index = g_gpuTimingNextSlot; + g_gpuTimingNextSlot = (g_gpuTimingNextSlot + 1) % kGpuTimingSlots; + auto& slot = g_gpuTimingSlots[index]; + { + std::lock_guard lock(g_gpuTimingMutex); + if (slot.reading && !slot.mapped) { + ++g_gpuTimingSkipped; // the GPU is more than a ring behind; leave this frame untimed + return; + } + if (slot.mapped) { + slot.readback.Unmap(); + slot.mapped = false; + } + slot.reading = false; + } + slot.pairs = 0; + slot.open = true; + g_gpuTimingCurrent = static_cast(index); +} + +const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept { + if (g_gpuTimingCurrent < 0) { + return nullptr; + } + auto& slot = g_gpuTimingSlots[static_cast(g_gpuTimingCurrent)]; + if (!slot.open || slot.pairs >= kGpuTimingPairs) { + return nullptr; + } + const uint32_t i = slot.pairs++; + slot.writes[i] = wgpu::PassTimestampWrites{ + .querySet = slot.querySet, + .beginningOfPassWriteIndex = 2 * i, + .endOfPassWriteIndex = 2 * i + 1, + }; + slot.categories[i] = category; + return &slot.writes[i]; +} + +void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept { + if (g_gpuTimingCurrent < 0) { + return; + } + auto& slot = g_gpuTimingSlots[static_cast(g_gpuTimingCurrent)]; + slot.open = false; + if (slot.pairs == 0) { + g_gpuTimingCurrent = -1; + return; + } + const uint32_t queries = 2 * slot.pairs; + encoder.ResolveQuerySet(slot.querySet, 0, queries, slot.resolve, 0); + encoder.CopyBufferToBuffer(slot.resolve, 0, slot.readback, 0, queries * sizeof(uint64_t)); +} + +void gpu_timing_after_submit() noexcept { + if (g_gpuTimingCurrent < 0) { + return; + } + const uint32_t index = static_cast(g_gpuTimingCurrent); + g_gpuTimingCurrent = -1; + auto& slot = g_gpuTimingSlots[index]; + const uint32_t pairs = slot.pairs; + { + std::lock_guard lock(g_gpuTimingMutex); + slot.reading = true; + slot.mapped = false; + } + slot.readback.MapAsync( + wgpu::MapMode::Read, 0, 2 * pairs * sizeof(uint64_t), wgpu::CallbackMode::AllowSpontaneous, + [index, pairs](wgpu::MapAsyncStatus status, wgpu::StringView) { + auto& slot = g_gpuTimingSlots[index]; + std::lock_guard lock(g_gpuTimingMutex); + if (status != wgpu::MapAsyncStatus::Success) { + slot.reading = false; + return; + } + const auto* stamps = + static_cast(slot.readback.GetConstMappedRange(0, 2 * pairs * sizeof(uint64_t))); + if (stamps != nullptr) { + uint64_t first = UINT64_MAX; + uint64_t last = 0; + for (uint32_t i = 0; i < pairs; ++i) { + const uint64_t begin = stamps[2 * i]; + const uint64_t end = stamps[2 * i + 1]; + if (end < begin) { + continue; + } + g_gpuTimingTotalsNs[static_cast(slot.categories[i])] += end - begin; + first = std::min(first, begin); + last = std::max(last, end); + } + if (last > first) { + g_gpuTimingSpanNs += last - first; + } + ++g_gpuTimingFrames; + } + slot.mapped = true; + }); +} + +std::string gpu_timing_report() { + std::lock_guard lock(g_gpuTimingMutex); + if (g_gpuTimingFrames == 0 && g_gpuTimingSkipped == 0) { + return {}; + } + static constexpr std::array(GpuTimingCategory::Count)> kNames{ + "mono", "eyeL", "eyeR", "interp", "screen", "panel", "efbcopy", "palette", "peek", "snapshot", "present"}; + std::string text; + if (g_gpuTimingFrames != 0) { + const double frames = g_gpuTimingFrames; + uint64_t sum = 0; + text += fmt::format("GPU ms/frame over {} frames: passes-span={:.2f}", g_gpuTimingFrames, + static_cast(g_gpuTimingSpanNs) / 1e6 / frames); + for (size_t i = 0; i < kNames.size(); ++i) { + if (g_gpuTimingTotalsNs[i] == 0) { + continue; + } + sum += g_gpuTimingTotalsNs[i]; + text += fmt::format(" {}={:.2f}", kNames[i], static_cast(g_gpuTimingTotalsNs[i]) / 1e6 / frames); + } + const uint64_t between = g_gpuTimingSpanNs > sum ? g_gpuTimingSpanNs - sum : 0; + text += fmt::format(" between-passes={:.2f}", static_cast(between) / 1e6 / frames); + } + if (g_gpuTimingSkipped != 0) { + text += fmt::format(" (untimed frames: {})", g_gpuTimingSkipped); + } + g_gpuTimingTotalsNs.fill(0); + g_gpuTimingSpanNs = 0; + g_gpuTimingFrames = 0; + g_gpuTimingSkipped = 0; + return text; +} + void after_submit() noexcept { depth_peek::after_submit(); efb_ram::after_submit(); diff --git a/aurora-main/lib/gfx/common.hpp b/aurora-main/lib/gfx/common.hpp index 613f3db..4612b68 100644 --- a/aurora-main/lib/gfx/common.hpp +++ b/aurora-main/lib/gfx/common.hpp @@ -8,6 +8,7 @@ #include #include #include +#include #include #include @@ -358,7 +359,14 @@ bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, c // Encode a sealed frame. Never touches the producer-visible recording state, // so this may run concurrently with the producer's FIFO drains. -void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true); +// `nativeRenderLastPass` limits the passes that do render work (texture bakes still run for every +// pass): a headset never shows an immersive frame's native render, so encode_sealed_frame stops it +// after the last pass whose EFB copy the eye replays sample. +void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true, + int32_t nativeRenderLastPass = INT32_MAX); +// Index of the last recorded pass that resolves an EFB copy other than the display copy, or -1 +// when no pass does: everything after it exists only for the presented image. +int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept; // Replays only main-EFB passes into one Aurora-owned eye target. Native // offscreen/EFB-copy passes are consumed from the mono render and are not @@ -414,6 +422,39 @@ bool is_offscreen() noexcept; uint32_t get_sample_count() noexcept; void clear_caches() noexcept; +// Per-pass GPU timing for the frame-rate log. When enabled and the device has TimestampQuery, +// every render or compute pass asks gpu_timing_pass() for timestamp writes under a category; the +// frame's queries are resolved into a small ring of readback buffers and the completed frames' +// durations are summed per category until gpu_timing_report() consumes them. Off by default: +// aurora.cpp enables it together with the Android frame-rate log. +enum class GpuTimingCategory : uint8_t { + Mono, // the native (desktop) render of the recorded GX passes + EyeLeft, // stereo replay of the left eye + EyeRight, // stereo replay of the right eye + Interpolated, // interpolated presentation slots + VirtualScreen, // the 2D virtual screen built for each eye + Panel, // the in-headset settings panel + EfbCopy, // EFB copy format conversions + Palette, // palette (TLUT) texture conversions + DepthPeek, // the depth snapshot compute pass + Snapshot, // presentation snapshot and its ImGui pass + Present, // the desktop presentation copy + Count, +}; +void gpu_timing_set_enabled(bool enabled) noexcept; +bool gpu_timing_enabled() noexcept; +// Opens the current frame's query slot; a frame whose slot is still being read back is skipped. +void gpu_timing_begin_frame() noexcept; +// Timestamp writes for one pass of the open frame, or nullptr when timing is off or exhausted. +const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept; +// Resolves the open frame's queries on `encoder`, which must be the frame's last submission. +void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept; +// After that submission: starts the readback of the resolved queries. +void gpu_timing_after_submit() noexcept; +// Per-frame averages of the frames read back since the last call, formatted for the log, or +// an empty string when nothing was measured. +std::string gpu_timing_report(); + namespace tex_palette_conv { struct ConvRequest; } // namespace tex_palette_conv diff --git a/aurora-main/lib/gfx/depth_peek.cpp b/aurora-main/lib/gfx/depth_peek.cpp index 66a7d5e..b488ef6 100644 --- a/aurora-main/lib/gfx/depth_peek.cpp +++ b/aurora-main/lib/gfx/depth_peek.cpp @@ -1,4 +1,5 @@ #include "depth_peek.hpp" +#include "common.hpp" #include "../dolphin/vi/vi_internal.hpp" #include "../gx/gx.hpp" @@ -403,6 +404,7 @@ void encode_frame_snapshot(const wgpu::CommandEncoder& cmd, const wgpu::TextureV const wgpu::ComputePassDescriptor passDescriptor{ .label = "Depth Peek Compute Pass", + .timestampWrites = gpu_timing_pass(GpuTimingCategory::DepthPeek), }; const auto pass = cmd.BeginComputePass(&passDescriptor); pass.SetPipeline(g_pipeline); diff --git a/aurora-main/lib/gfx/tex_copy_conv.cpp b/aurora-main/lib/gfx/tex_copy_conv.cpp index 6b68864..4cd4aa9 100644 --- a/aurora-main/lib/gfx/tex_copy_conv.cpp +++ b/aurora-main/lib/gfx/tex_copy_conv.cpp @@ -1,4 +1,5 @@ #include "tex_copy_conv.hpp" +#include "common.hpp" #include "tex_copy_format_contract.hpp" #include "../internal.hpp" @@ -586,6 +587,7 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con .label = "TexCopyConv Pass", .colorAttachmentCount = colorAttachments.size(), .colorAttachments = colorAttachments.data(), + .timestampWrites = gpu_timing_pass(GpuTimingCategory::EfbCopy), }; const auto pass = cmd.BeginRenderPass(&renderPassDescriptor); pass.SetPipeline(pipeline); diff --git a/aurora-main/lib/gfx/tex_palette_conv.cpp b/aurora-main/lib/gfx/tex_palette_conv.cpp index 842cca3..38936f8 100644 --- a/aurora-main/lib/gfx/tex_palette_conv.cpp +++ b/aurora-main/lib/gfx/tex_palette_conv.cpp @@ -1,4 +1,5 @@ #include "tex_palette_conv.hpp" +#include "common.hpp" #include "../internal.hpp" #include "../webgpu/gpu.hpp" @@ -249,6 +250,7 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) { .label = "TexPaletteConv Pass", .colorAttachmentCount = colorAttachments.size(), .colorAttachments = colorAttachments.data(), + .timestampWrites = gpu_timing_pass(GpuTimingCategory::Palette), }; const auto pass = cmd.BeginRenderPass(&renderPassDescriptor); pass.SetPipeline(pipeline); diff --git a/aurora-main/lib/stereo_overlay.cpp b/aurora-main/lib/stereo_overlay.cpp index 1f0a9f1..4395d69 100644 --- a/aurora-main/lib/stereo_overlay.cpp +++ b/aurora-main/lib/stereo_overlay.cpp @@ -239,6 +239,7 @@ void composite(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& tar .label = eyeIndex == 0 ? "Headset panel left eye" : "Headset panel right eye", .colorAttachmentCount = attachments.size(), .colorAttachments = attachments.data(), + .timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel), }; const auto pass = encoder.BeginRenderPass(&descriptor); pass.SetPipeline(state.pipeline); @@ -294,6 +295,7 @@ wgpu::CommandBuffer prepare(ImDrawData* drawData, float widthFraction) noexcept .label = "Headset panel ImGui pass", .colorAttachmentCount = attachments.size(), .colorAttachments = attachments.data(), + .timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel), }; bool drawn = false; { diff --git a/aurora-main/lib/webgpu/gpu.cpp b/aurora-main/lib/webgpu/gpu.cpp index 58ed9ec..b421eb8 100644 --- a/aurora-main/lib/webgpu/gpu.cpp +++ b/aurora-main/lib/webgpu/gpu.cpp @@ -80,6 +80,7 @@ wgpu::Instance g_instance; static wgpu::AdapterInfo g_adapterInfo; static wgpu::SurfaceCapabilities g_surfaceCapabilities; bool g_bcTexturesSupported; +bool g_timestampQueriesSupported = false; // Written by Dawn's device-loss callback and consumed at ordered frame boundaries. Keep the // callback free of logging, allocation, teardown and renderer state mutation. static std::atomic_bool g_deviceLost{false}; @@ -702,6 +703,12 @@ bool initialize(AuroraBackend auroraBackend) { g_bcTexturesSupported = true; requiredFeatures.push_back(feature); } + // Per-pass GPU timing for the frame-rate log (gfx::gpu_timing_*). Requesting the feature + // costs nothing until a pass carries timestamp writes. + if (feature == wgpu::FeatureName::TimestampQuery) { + g_timestampQueriesSupported = true; + requiredFeatures.push_back(feature); + } // The presenter calls device and queue methods while the frame worker encodes, which Dawn only // supports with this feature; without it the two race inside the device's dynamic uploader. if (feature == wgpu::FeatureName::ImplicitDeviceSynchronization) { @@ -769,10 +776,15 @@ bool initialize(AuroraBackend auroraBackend) { if (g_backendType == wgpu::BackendType::Vulkan) { enableToggles.push_back("vulkan_monolithic_pipeline_cache"); } + // Dawn quantizes timestamp queries to 100 us for web privacy; the per-pass GPU timing wants + // the raw values. + const std::array disableToggles{"timestamp_quantization"}; const wgpu::DawnTogglesDescriptor togglesDescriptor({ .nextInChain = &cacheDescriptor, .enabledToggleCount = enableToggles.size(), .enabledToggles = enableToggles.data(), + .disabledToggleCount = g_timestampQueriesSupported ? disableToggles.size() : 0, + .disabledToggles = disableToggles.data(), }); #endif wgpu::DeviceDescriptor deviceDescriptor; diff --git a/aurora-main/lib/webgpu/gpu.hpp b/aurora-main/lib/webgpu/gpu.hpp index 7b96e95..eb2f1a4 100644 --- a/aurora-main/lib/webgpu/gpu.hpp +++ b/aurora-main/lib/webgpu/gpu.hpp @@ -58,6 +58,8 @@ extern wgpu::RenderPipeline g_CopyPipeline; extern wgpu::BindGroup g_CopyBindGroup; extern wgpu::Instance g_instance; extern bool g_bcTexturesSupported; +// The device was created with TimestampQuery, so passes may carry timestamp writes (gfx::gpu_timing_*). +extern bool g_timestampQueriesSupported; bool initialize(AuroraBackend backend); void shutdown(); diff --git a/docs/quest-port.md b/docs/quest-port.md index fbde6a3..eccd422 100644 --- a/docs/quest-port.md +++ b/docs/quest-port.md @@ -143,6 +143,12 @@ suggested for `oculus/touch_controller` and `khr/simple_controller`. provider is registered on Android, Aurora skips the surface present and the desktop mirror copy (`headset_owns_display` in `lib/aurora.cpp`). The game's own render size is unaffected: at `resolution_multiplier = 1` it is 640x528. + Since 2026-09-19 an immersive race also stops that native render after the + last pass whose EFB copy the eye replays sample (`last_pass_feeding_replay`): + the main scene and display copy of a 1280x720 image nobody sees were 4 to + 6 ms of a 12 ms GPU frame on a Quest 3. A pending CPU readback of an EFB + copy or a frame capture still renders the whole image, and menus (the + virtual screen) keep it because their eyes are built from that snapshot. - **JNI only on the real thread stack.** Guest threads run on libco stacks inside the SDL thread, and SDL's Android event pump can reach Java (joystick polling, HIDAPI). ART binds JNI transitions to the thread's real stack, so @@ -582,7 +588,7 @@ the app: | `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update | | `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds | | `debug.wiicompiled.inject :