Implement per-pass GPU timing for performance tracking and optimization & fixed mono in Immersive

This commit is contained in:
iChris4 committed 2026-09-19 19:23:57 +02:00
1 parent f0e5e43985
commit 47b59c7294
10 files changed
+344 -5

No files matched your search

+30 -2
View File
@@ -756,6 +756,7 @@ void encode_virtual_screen_eye(wgpu::CommandEncoder& encoder, const webgpu::Pres
.label = eyeIndex == 0 ? "Virtual screen left eye" : "Virtual screen right eye", .label = eyeIndex == 0 ? "Virtual screen left eye" : "Virtual screen right eye",
.colorAttachmentCount = attachments.size(), .colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(), .colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::VirtualScreen),
}; };
{ {
const auto pass = encoder.BeginRenderPass(&descriptor); const auto pass = encoder.BeginRenderPass(&descriptor);
@@ -1274,6 +1275,7 @@ bool present_presentation_job(const PresentationJob& job) {
.label = "Presentation copy pass", .label = "Presentation copy pass",
.colorAttachmentCount = attachments.size(), .colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(), .colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Present),
}; };
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor); const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(webgpu::g_CopyPipeline); pass.SetPipeline(webgpu::g_CopyPipeline);
@@ -1576,6 +1578,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
.label = "Interpolation snapshot pass", .label = "Interpolation snapshot pass",
.colorAttachmentCount = attachments.size(), .colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(), .colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
}; };
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor); const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
const auto imageWidth = static_cast<float>(image.texture.size.width); const auto imageWidth = static_cast<float>(image.texture.size.width);
@@ -1627,6 +1630,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
.label = "Snapshot ImGui pass", .label = "Snapshot ImGui pass",
.colorAttachmentCount = attachments.size(), .colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(), .colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
}; };
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor); const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
pass.SetViewport(0.f, 0.f, static_cast<float>(image.texture.size.width), pass.SetViewport(0.f, 0.f, static_cast<float>(image.texture.size.width),
@@ -1829,6 +1833,9 @@ void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept {
void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag, void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag,
const StereoSceneAnchor& sceneAnchor) { const StereoSceneAnchor& sceneAnchor) {
ZoneScopedN("Seal frame"); ZoneScopedN("Seal frame");
// Every pass this cycle encodes, from the seal's probe blits to the final eye, is timed under
// one frame; encode_sealed_frame resolves it on its last submission.
gfx::gpu_timing_begin_frame();
const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{ const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{
.label = "Redraw encoder", .label = "Redraw encoder",
}; };
@@ -1988,7 +1995,18 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
// A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed // A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed
// stream would mutate an already-rendered EFB. Render once, then duplicate into the slots. // stream would mutate an already-rendered EFB. Render once, then duplicate into the slots.
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo); //
// On a headset an immersive frame's native render is never presented: the eyes replay the draws
// themselves and only sample the EFB copies it resolves. So it stops after the last pass that
// produces one of those copies (never the display copy), which on a Quest 3 was 4 to 6 ms of a
// 12 ms GPU frame spent on a 1280x720 image nobody saw. A pending CPU readback or a frame
// capture still gets the whole image.
int32_t nativeRenderLastPass = INT32_MAX;
if (headsetOnly && immersiveReplay && !gfx::efb_ram::has_pending() &&
g_captureFrame.load(std::memory_order_acquire) == UINT32_MAX) {
nativeRenderLastPass = gfx::last_pass_feeding_replay(sealedFrame);
}
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo, nativeRenderLastPass);
// The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder; // The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder;
// completion is harvested in gfx::after_submit, never waited on here. // completion is harvested in gfx::after_submit, never waited on here.
gfx::efb_ram::encode_async_downloads(encoder); gfx::efb_ram::encode_async_downloads(encoder);
@@ -2070,7 +2088,9 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
.presentAt = slotPresentDeadline(ctx.interpolatedFrameCount), .presentAt = slotPresentDeadline(ctx.interpolatedFrameCount),
.interpolated = false, .interpolated = false,
}); });
gfx::gpu_timing_end_frame(encoder);
submitEncodedSlot(encoder, pendingStereoSink ? &*pendingStereoSink : nullptr); submitEncodedSlot(encoder, pendingStereoSink ? &*pendingStereoSink : nullptr);
gfx::gpu_timing_after_submit();
// A group that finished encoding past its anchor slides forward by whole display periods, never // A group that finished encoding past its anchor slides forward by whole display periods, never
// per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period. // per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period.
@@ -2180,7 +2200,12 @@ void record_frame_telemetry() {
{ {
// `adb shell setprop debug.wiicompiled.fpslog 1` before launch logs the game's rendered frame rate every five // `adb shell setprop debug.wiicompiled.fpslog 1` before launch logs the game's rendered frame rate every five
// seconds. The headset compositor's own log (logcat tag VrApi) repeats frames, so it cannot show this. // seconds. The headset compositor's own log (logcat tag VrApi) repeats frames, so it cannot show this.
static const bool fpsLog = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1; static const bool fpsLog = [] {
const bool on = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
// The same switch turns on the per-pass GPU timestamps reported below the frame-rate line.
gfx::gpu_timing_set_enabled(on);
return on;
}();
if (fpsLog) { if (fpsLog) {
static auto windowStart = std::chrono::steady_clock::now(); static auto windowStart = std::chrono::steady_clock::now();
static uint32_t windowFrames = 0; static uint32_t windowFrames = 0;
@@ -2203,6 +2228,9 @@ void record_frame_telemetry() {
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding", "prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding",
windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal, windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal,
permitWait, prepare, encode); permitWait, prepare, encode);
if (const std::string gpuTiming = gfx::gpu_timing_report(); !gpuTiming.empty()) {
Log.info("{}", gpuTiming);
}
windowStart = now; windowStart = now;
windowFrames = 0; windowFrames = 0;
} }
+230 -1
View File
@@ -1724,6 +1724,9 @@ struct RenderInvocation {
uint32_t localPlayerCount = 1; uint32_t localPlayerCount = 1;
// Inclusive index of the last pass to replay; -1 replays every pass. // Inclusive index of the last pass to replay; -1 replays every pass.
int32_t replayLastPass = -1; int32_t replayLastPass = -1;
// Inclusive index of the last pass that does render work; texture bakes still run for the
// passes after it. See last_pass_feeding_replay.
int32_t renderLastPass = INT32_MAX;
bool finalize = true; bool finalize = true;
bool replayOnlyEfb = false; bool replayOnlyEfb = false;
bool skipCopyClears = false; bool skipCopyClears = false;
@@ -1755,6 +1758,11 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
tex_palette_conv::run(cmd, conv); tex_palette_conv::run(cmd, conv);
} }
} }
if (static_cast<int32_t>(i) > invocation.renderLastPass) {
// Nothing after the last replay-feeding resolve is shown or sampled on a headset; the
// bakes above are all these passes owe the eye replays.
continue;
}
const bool hasRenderWork = passInfo.clearColor || passInfo.clearDepth || !passInfo.commands.empty(); const bool hasRenderWork = passInfo.clearColor || passInfo.clearDepth || !passInfo.commands.empty();
if (i == renderPasses.size() - 1) { if (i == renderPasses.size() - 1) {
ASSERT(!passInfo.resolveTarget, "Final render pass must not have resolve target"); ASSERT(!passInfo.resolveTarget, "Final render pass must not have resolve target");
@@ -1793,11 +1801,16 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
.depthStoreOp = wgpu::StoreOp::Store, .depthStoreOp = wgpu::StoreOp::Store,
.depthClearValue = passInfo.clearDepthValue, .depthClearValue = passInfo.clearDepthValue,
}; };
const GpuTimingCategory timingCategory = invocation.stereoEye == 0 ? GpuTimingCategory::EyeLeft
: invocation.stereoEye == 1 ? GpuTimingCategory::EyeRight
: invocation.interpolatedFrame >= 0 ? GpuTimingCategory::Interpolated
: GpuTimingCategory::Mono;
const wgpu::RenderPassDescriptor renderPassDescriptor{ const wgpu::RenderPassDescriptor renderPassDescriptor{
.label = render_pass_label(i), .label = render_pass_label(i),
.colorAttachmentCount = attachments.size(), .colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(), .colorAttachments = attachments.data(),
.depthStencilAttachment = &depthStencilAttachment, .depthStencilAttachment = &depthStencilAttachment,
.timestampWrites = gpu_timing_pass(timingCategory),
}; };
auto pass = cmd.BeginRenderPass(&renderPassDescriptor); auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
@@ -1908,15 +1921,28 @@ void seal_frame(SealedFrame& out) noexcept {
g_currentRenderPass = UINT32_MAX; g_currentRenderPass = UINT32_MAX;
} }
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) { void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize,
int32_t nativeRenderLastPass) {
render_impl(frame.data().passes, cmd, render_impl(frame.data().passes, cmd,
RenderInvocation{ RenderInvocation{
.interpolatedFrame = interpolatedFrame, .interpolatedFrame = interpolatedFrame,
.renderLastPass = nativeRenderLastPass,
.finalize = finalize, .finalize = finalize,
.encodeTextureBakes = interpolatedFrame < 0, .encodeTextureBakes = interpolatedFrame < 0,
}); });
} }
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept {
const auto& passes = frame.data().passes;
int32_t last = -1;
for (size_t i = 0; i < passes.size(); ++i) {
if (passes[i].resolveTarget && !passes[i].displayCopyResolve) {
last = static_cast<int32_t>(i);
}
}
return last;
}
bool has_late_stereo_replay(const SealedFrame& frame) noexcept { bool has_late_stereo_replay(const SealedFrame& frame) noexcept {
const auto& data = frame.data().stereo; const auto& data = frame.data().stereo;
return data.generation != 0 && data.generation == g_replayBufferGeneration.load(std::memory_order_acquire) && return data.generation != 0 && data.generation == g_replayBufferGeneration.load(std::memory_order_acquire) &&
@@ -2028,6 +2054,209 @@ void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize)
} }
} }
// --- Per-pass GPU timing (see common.hpp) -------------------------------------------------------
namespace {
constexpr uint32_t kGpuTimingSlots = 4;
constexpr uint32_t kGpuTimingPairs = 62;
constexpr uint32_t kGpuTimingQueries = 2 * kGpuTimingPairs;
struct GpuTimingSlot {
wgpu::QuerySet querySet;
wgpu::Buffer resolve;
wgpu::Buffer readback;
std::array<wgpu::PassTimestampWrites, kGpuTimingPairs> writes{};
std::array<GpuTimingCategory, kGpuTimingPairs> categories{};
uint32_t pairs = 0;
bool open = false; // between the frame's begin and end
bool reading = false; // readback in flight or mapped
bool mapped = false; // the callback ran; the encoding thread unmaps on reuse
};
std::atomic<bool> g_gpuTimingEnabled{false};
std::array<GpuTimingSlot, kGpuTimingSlots> g_gpuTimingSlots;
uint32_t g_gpuTimingNextSlot = 0;
int32_t g_gpuTimingCurrent = -1;
bool g_gpuTimingReady = false;
// Guards the totals below and every slot's reading/mapped flags: the map callback may run on
// whichever thread processes Dawn's events.
std::mutex g_gpuTimingMutex;
std::array<uint64_t, static_cast<size_t>(GpuTimingCategory::Count)> g_gpuTimingTotalsNs{};
uint64_t g_gpuTimingSpanNs = 0;
uint32_t g_gpuTimingFrames = 0;
uint32_t g_gpuTimingSkipped = 0;
bool gpu_timing_create_slots() {
if (g_gpuTimingReady) {
return true;
}
if (!webgpu::g_timestampQueriesSupported || !webgpu::g_device) {
return false;
}
for (auto& slot : g_gpuTimingSlots) {
const wgpu::QuerySetDescriptor querySetDescriptor{
.label = "GPU timing queries",
.type = wgpu::QueryType::Timestamp,
.count = kGpuTimingQueries,
};
slot.querySet = webgpu::g_device.CreateQuerySet(&querySetDescriptor);
const wgpu::BufferDescriptor resolveDescriptor{
.label = "GPU timing resolve",
.usage = wgpu::BufferUsage::QueryResolve | wgpu::BufferUsage::CopySrc,
.size = kGpuTimingQueries * sizeof(uint64_t),
};
slot.resolve = webgpu::g_device.CreateBuffer(&resolveDescriptor);
const wgpu::BufferDescriptor readbackDescriptor{
.label = "GPU timing readback",
.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst,
.size = kGpuTimingQueries * sizeof(uint64_t),
};
slot.readback = webgpu::g_device.CreateBuffer(&readbackDescriptor);
}
g_gpuTimingReady = true;
return true;
}
} // namespace
void gpu_timing_set_enabled(bool enabled) noexcept { g_gpuTimingEnabled.store(enabled, std::memory_order_relaxed); }
bool gpu_timing_enabled() noexcept { return g_gpuTimingEnabled.load(std::memory_order_relaxed); }
void gpu_timing_begin_frame() noexcept {
g_gpuTimingCurrent = -1;
if (!gpu_timing_enabled() || !gpu_timing_create_slots()) {
return;
}
const uint32_t index = g_gpuTimingNextSlot;
g_gpuTimingNextSlot = (g_gpuTimingNextSlot + 1) % kGpuTimingSlots;
auto& slot = g_gpuTimingSlots[index];
{
std::lock_guard lock(g_gpuTimingMutex);
if (slot.reading && !slot.mapped) {
++g_gpuTimingSkipped; // the GPU is more than a ring behind; leave this frame untimed
return;
}
if (slot.mapped) {
slot.readback.Unmap();
slot.mapped = false;
}
slot.reading = false;
}
slot.pairs = 0;
slot.open = true;
g_gpuTimingCurrent = static_cast<int32_t>(index);
}
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept {
if (g_gpuTimingCurrent < 0) {
return nullptr;
}
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
if (!slot.open || slot.pairs >= kGpuTimingPairs) {
return nullptr;
}
const uint32_t i = slot.pairs++;
slot.writes[i] = wgpu::PassTimestampWrites{
.querySet = slot.querySet,
.beginningOfPassWriteIndex = 2 * i,
.endOfPassWriteIndex = 2 * i + 1,
};
slot.categories[i] = category;
return &slot.writes[i];
}
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept {
if (g_gpuTimingCurrent < 0) {
return;
}
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
slot.open = false;
if (slot.pairs == 0) {
g_gpuTimingCurrent = -1;
return;
}
const uint32_t queries = 2 * slot.pairs;
encoder.ResolveQuerySet(slot.querySet, 0, queries, slot.resolve, 0);
encoder.CopyBufferToBuffer(slot.resolve, 0, slot.readback, 0, queries * sizeof(uint64_t));
}
void gpu_timing_after_submit() noexcept {
if (g_gpuTimingCurrent < 0) {
return;
}
const uint32_t index = static_cast<uint32_t>(g_gpuTimingCurrent);
g_gpuTimingCurrent = -1;
auto& slot = g_gpuTimingSlots[index];
const uint32_t pairs = slot.pairs;
{
std::lock_guard lock(g_gpuTimingMutex);
slot.reading = true;
slot.mapped = false;
}
slot.readback.MapAsync(
wgpu::MapMode::Read, 0, 2 * pairs * sizeof(uint64_t), wgpu::CallbackMode::AllowSpontaneous,
[index, pairs](wgpu::MapAsyncStatus status, wgpu::StringView) {
auto& slot = g_gpuTimingSlots[index];
std::lock_guard lock(g_gpuTimingMutex);
if (status != wgpu::MapAsyncStatus::Success) {
slot.reading = false;
return;
}
const auto* stamps =
static_cast<const uint64_t*>(slot.readback.GetConstMappedRange(0, 2 * pairs * sizeof(uint64_t)));
if (stamps != nullptr) {
uint64_t first = UINT64_MAX;
uint64_t last = 0;
for (uint32_t i = 0; i < pairs; ++i) {
const uint64_t begin = stamps[2 * i];
const uint64_t end = stamps[2 * i + 1];
if (end < begin) {
continue;
}
g_gpuTimingTotalsNs[static_cast<size_t>(slot.categories[i])] += end - begin;
first = std::min(first, begin);
last = std::max(last, end);
}
if (last > first) {
g_gpuTimingSpanNs += last - first;
}
++g_gpuTimingFrames;
}
slot.mapped = true;
});
}
std::string gpu_timing_report() {
std::lock_guard lock(g_gpuTimingMutex);
if (g_gpuTimingFrames == 0 && g_gpuTimingSkipped == 0) {
return {};
}
static constexpr std::array<const char*, static_cast<size_t>(GpuTimingCategory::Count)> kNames{
"mono", "eyeL", "eyeR", "interp", "screen", "panel", "efbcopy", "palette", "peek", "snapshot", "present"};
std::string text;
if (g_gpuTimingFrames != 0) {
const double frames = g_gpuTimingFrames;
uint64_t sum = 0;
text += fmt::format("GPU ms/frame over {} frames: passes-span={:.2f}", g_gpuTimingFrames,
static_cast<double>(g_gpuTimingSpanNs) / 1e6 / frames);
for (size_t i = 0; i < kNames.size(); ++i) {
if (g_gpuTimingTotalsNs[i] == 0) {
continue;
}
sum += g_gpuTimingTotalsNs[i];
text += fmt::format(" {}={:.2f}", kNames[i], static_cast<double>(g_gpuTimingTotalsNs[i]) / 1e6 / frames);
}
const uint64_t between = g_gpuTimingSpanNs > sum ? g_gpuTimingSpanNs - sum : 0;
text += fmt::format(" between-passes={:.2f}", static_cast<double>(between) / 1e6 / frames);
}
if (g_gpuTimingSkipped != 0) {
text += fmt::format(" (untimed frames: {})", g_gpuTimingSkipped);
}
g_gpuTimingTotalsNs.fill(0);
g_gpuTimingSpanNs = 0;
g_gpuTimingFrames = 0;
g_gpuTimingSkipped = 0;
return text;
}
void after_submit() noexcept { void after_submit() noexcept {
depth_peek::after_submit(); depth_peek::after_submit();
efb_ram::after_submit(); efb_ram::after_submit();
+42 -1
View File
@@ -8,6 +8,7 @@
#include <cstring> #include <cstring>
#include <array> #include <array>
#include <memory> #include <memory>
#include <string>
#include <type_traits> #include <type_traits>
#include <utility> #include <utility>
@@ -358,7 +359,14 @@ bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, c
// Encode a sealed frame. Never touches the producer-visible recording state, // Encode a sealed frame. Never touches the producer-visible recording state,
// so this may run concurrently with the producer's FIFO drains. // so this may run concurrently with the producer's FIFO drains.
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true); // `nativeRenderLastPass` limits the passes that do render work (texture bakes still run for every
// pass): a headset never shows an immersive frame's native render, so encode_sealed_frame stops it
// after the last pass whose EFB copy the eye replays sample.
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true,
int32_t nativeRenderLastPass = INT32_MAX);
// Index of the last recorded pass that resolves an EFB copy other than the display copy, or -1
// when no pass does: everything after it exists only for the presented image.
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept;
// Replays only main-EFB passes into one Aurora-owned eye target. Native // Replays only main-EFB passes into one Aurora-owned eye target. Native
// offscreen/EFB-copy passes are consumed from the mono render and are not // offscreen/EFB-copy passes are consumed from the mono render and are not
@@ -414,6 +422,39 @@ bool is_offscreen() noexcept;
uint32_t get_sample_count() noexcept; uint32_t get_sample_count() noexcept;
void clear_caches() noexcept; void clear_caches() noexcept;
// Per-pass GPU timing for the frame-rate log. When enabled and the device has TimestampQuery,
// every render or compute pass asks gpu_timing_pass() for timestamp writes under a category; the
// frame's queries are resolved into a small ring of readback buffers and the completed frames'
// durations are summed per category until gpu_timing_report() consumes them. Off by default:
// aurora.cpp enables it together with the Android frame-rate log.
enum class GpuTimingCategory : uint8_t {
Mono, // the native (desktop) render of the recorded GX passes
EyeLeft, // stereo replay of the left eye
EyeRight, // stereo replay of the right eye
Interpolated, // interpolated presentation slots
VirtualScreen, // the 2D virtual screen built for each eye
Panel, // the in-headset settings panel
EfbCopy, // EFB copy format conversions
Palette, // palette (TLUT) texture conversions
DepthPeek, // the depth snapshot compute pass
Snapshot, // presentation snapshot and its ImGui pass
Present, // the desktop presentation copy
Count,
};
void gpu_timing_set_enabled(bool enabled) noexcept;
bool gpu_timing_enabled() noexcept;
// Opens the current frame's query slot; a frame whose slot is still being read back is skipped.
void gpu_timing_begin_frame() noexcept;
// Timestamp writes for one pass of the open frame, or nullptr when timing is off or exhausted.
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept;
// Resolves the open frame's queries on `encoder`, which must be the frame's last submission.
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept;
// After that submission: starts the readback of the resolved queries.
void gpu_timing_after_submit() noexcept;
// Per-frame averages of the frames read back since the last call, formatted for the log, or
// an empty string when nothing was measured.
std::string gpu_timing_report();
namespace tex_palette_conv { namespace tex_palette_conv {
struct ConvRequest; struct ConvRequest;
} // namespace tex_palette_conv } // namespace tex_palette_conv
+2
View File
@@ -1,4 +1,5 @@
#include "depth_peek.hpp" #include "depth_peek.hpp"
#include "common.hpp"
#include "../dolphin/vi/vi_internal.hpp" #include "../dolphin/vi/vi_internal.hpp"
#include "../gx/gx.hpp" #include "../gx/gx.hpp"
@@ -403,6 +404,7 @@ void encode_frame_snapshot(const wgpu::CommandEncoder& cmd, const wgpu::TextureV
const wgpu::ComputePassDescriptor passDescriptor{ const wgpu::ComputePassDescriptor passDescriptor{
.label = "Depth Peek Compute Pass", .label = "Depth Peek Compute Pass",
.timestampWrites = gpu_timing_pass(GpuTimingCategory::DepthPeek),
}; };
const auto pass = cmd.BeginComputePass(&passDescriptor); const auto pass = cmd.BeginComputePass(&passDescriptor);
pass.SetPipeline(g_pipeline); pass.SetPipeline(g_pipeline);
+2
View File
@@ -1,4 +1,5 @@
#include "tex_copy_conv.hpp" #include "tex_copy_conv.hpp"
#include "common.hpp"
#include "tex_copy_format_contract.hpp" #include "tex_copy_format_contract.hpp"
#include "../internal.hpp" #include "../internal.hpp"
@@ -586,6 +587,7 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con
.label = "TexCopyConv Pass", .label = "TexCopyConv Pass",
.colorAttachmentCount = colorAttachments.size(), .colorAttachmentCount = colorAttachments.size(),
.colorAttachments = colorAttachments.data(), .colorAttachments = colorAttachments.data(),
.timestampWrites = gpu_timing_pass(GpuTimingCategory::EfbCopy),
}; };
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor); const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(pipeline); pass.SetPipeline(pipeline);
+2
View File
@@ -1,4 +1,5 @@
#include "tex_palette_conv.hpp" #include "tex_palette_conv.hpp"
#include "common.hpp"
#include "../internal.hpp" #include "../internal.hpp"
#include "../webgpu/gpu.hpp" #include "../webgpu/gpu.hpp"
@@ -249,6 +250,7 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
.label = "TexPaletteConv Pass", .label = "TexPaletteConv Pass",
.colorAttachmentCount = colorAttachments.size(), .colorAttachmentCount = colorAttachments.size(),
.colorAttachments = colorAttachments.data(), .colorAttachments = colorAttachments.data(),
.timestampWrites = gpu_timing_pass(GpuTimingCategory::Palette),
}; };
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor); const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(pipeline); pass.SetPipeline(pipeline);
+2
View File
@@ -239,6 +239,7 @@ void composite(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& tar
.label = eyeIndex == 0 ? "Headset panel left eye" : "Headset panel right eye", .label = eyeIndex == 0 ? "Headset panel left eye" : "Headset panel right eye",
.colorAttachmentCount = attachments.size(), .colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(), .colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
}; };
const auto pass = encoder.BeginRenderPass(&descriptor); const auto pass = encoder.BeginRenderPass(&descriptor);
pass.SetPipeline(state.pipeline); pass.SetPipeline(state.pipeline);
@@ -294,6 +295,7 @@ wgpu::CommandBuffer prepare(ImDrawData* drawData, float widthFraction) noexcept
.label = "Headset panel ImGui pass", .label = "Headset panel ImGui pass",
.colorAttachmentCount = attachments.size(), .colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(), .colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
}; };
bool drawn = false; bool drawn = false;
{ {
+12
View File
@@ -80,6 +80,7 @@ wgpu::Instance g_instance;
static wgpu::AdapterInfo g_adapterInfo; static wgpu::AdapterInfo g_adapterInfo;
static wgpu::SurfaceCapabilities g_surfaceCapabilities; static wgpu::SurfaceCapabilities g_surfaceCapabilities;
bool g_bcTexturesSupported; bool g_bcTexturesSupported;
bool g_timestampQueriesSupported = false;
// Written by Dawn's device-loss callback and consumed at ordered frame boundaries. Keep the // Written by Dawn's device-loss callback and consumed at ordered frame boundaries. Keep the
// callback free of logging, allocation, teardown and renderer state mutation. // callback free of logging, allocation, teardown and renderer state mutation.
static std::atomic_bool g_deviceLost{false}; static std::atomic_bool g_deviceLost{false};
@@ -702,6 +703,12 @@ bool initialize(AuroraBackend auroraBackend) {
g_bcTexturesSupported = true; g_bcTexturesSupported = true;
requiredFeatures.push_back(feature); requiredFeatures.push_back(feature);
} }
// Per-pass GPU timing for the frame-rate log (gfx::gpu_timing_*). Requesting the feature
// costs nothing until a pass carries timestamp writes.
if (feature == wgpu::FeatureName::TimestampQuery) {
g_timestampQueriesSupported = true;
requiredFeatures.push_back(feature);
}
// The presenter calls device and queue methods while the frame worker encodes, which Dawn only // The presenter calls device and queue methods while the frame worker encodes, which Dawn only
// supports with this feature; without it the two race inside the device's dynamic uploader. // supports with this feature; without it the two race inside the device's dynamic uploader.
if (feature == wgpu::FeatureName::ImplicitDeviceSynchronization) { if (feature == wgpu::FeatureName::ImplicitDeviceSynchronization) {
@@ -769,10 +776,15 @@ bool initialize(AuroraBackend auroraBackend) {
if (g_backendType == wgpu::BackendType::Vulkan) { if (g_backendType == wgpu::BackendType::Vulkan) {
enableToggles.push_back("vulkan_monolithic_pipeline_cache"); enableToggles.push_back("vulkan_monolithic_pipeline_cache");
} }
// Dawn quantizes timestamp queries to 100 us for web privacy; the per-pass GPU timing wants
// the raw values.
const std::array<const char*, 1> disableToggles{"timestamp_quantization"};
const wgpu::DawnTogglesDescriptor togglesDescriptor({ const wgpu::DawnTogglesDescriptor togglesDescriptor({
.nextInChain = &cacheDescriptor, .nextInChain = &cacheDescriptor,
.enabledToggleCount = enableToggles.size(), .enabledToggleCount = enableToggles.size(),
.enabledToggles = enableToggles.data(), .enabledToggles = enableToggles.data(),
.disabledToggleCount = g_timestampQueriesSupported ? disableToggles.size() : 0,
.disabledToggles = disableToggles.data(),
}); });
#endif #endif
wgpu::DeviceDescriptor deviceDescriptor; wgpu::DeviceDescriptor deviceDescriptor;
+2
View File
@@ -58,6 +58,8 @@ extern wgpu::RenderPipeline g_CopyPipeline;
extern wgpu::BindGroup g_CopyBindGroup; extern wgpu::BindGroup g_CopyBindGroup;
extern wgpu::Instance g_instance; extern wgpu::Instance g_instance;
extern bool g_bcTexturesSupported; extern bool g_bcTexturesSupported;
// The device was created with TimestampQuery, so passes may carry timestamp writes (gfx::gpu_timing_*).
extern bool g_timestampQueriesSupported;
bool initialize(AuroraBackend backend); bool initialize(AuroraBackend backend);
void shutdown(); void shutdown();
+20 -1
View File
@@ -143,6 +143,12 @@ suggested for `oculus/touch_controller` and `khr/simple_controller`.
provider is registered on Android, Aurora skips the surface present and the provider is registered on Android, Aurora skips the surface present and the
desktop mirror copy (`headset_owns_display` in `lib/aurora.cpp`). The game's desktop mirror copy (`headset_owns_display` in `lib/aurora.cpp`). The game's
own render size is unaffected: at `resolution_multiplier = 1` it is 640x528. own render size is unaffected: at `resolution_multiplier = 1` it is 640x528.
Since 2026-09-19 an immersive race also stops that native render after the
last pass whose EFB copy the eye replays sample (`last_pass_feeding_replay`):
the main scene and display copy of a 1280x720 image nobody sees were 4 to
6 ms of a 12 ms GPU frame on a Quest 3. A pending CPU readback of an EFB
copy or a frame capture still renders the whole image, and menus (the
virtual screen) keep it because their eyes are built from that snapshot.
- **JNI only on the real thread stack.** Guest threads run on libco stacks - **JNI only on the real thread stack.** Guest threads run on libco stacks
inside the SDL thread, and SDL's Android event pump can reach Java (joystick inside the SDL thread, and SDL's Android event pump can reach Java (joystick
polling, HIDAPI). ART binds JNI transitions to the thread's real stack, so polling, HIDAPI). ART binds JNI transitions to the thread's real stack, so
@@ -582,7 +588,7 @@ the app:
| `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update | | `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update |
| `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds | | `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds |
| `debug.wiicompiled.inject <n>:<button>` | Presses `a`, `b`, `x`, `y`, `start`, `up`, `down`, `left` or `right` for 12 XR frames each time `<n>` changes. As a Wii Remote, `x`/`y`/`start` are 1/2/+, the directions push the Nunchuk stick, and `home`, `c` and `z` also exist. `panel` presses the settings panel's button (left Y, or both thumbsticks as a gamepad), opening or closing it (see `OPENXR.md`) | | `debug.wiicompiled.inject <n>:<button>` | Presses `a`, `b`, `x`, `y`, `start`, `up`, `down`, `left` or `right` for 12 XR frames each time `<n>` changes. As a Wii Remote, `x`/`y`/`start` are 1/2/+, the directions push the Nunchuk stick, and `home`, `c` and `z` also exist. `panel` presses the settings panel's button (left Y, or both thumbsticks as a gamepad), opening or closing it (see `OPENXR.md`) |
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) | | `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. A second line gives the GPU time per frame from timestamp queries on every pass (`mono` native render, `eyeL`/`eyeR` replays, `screen`, `panel`, `efbcopy`, `palette`, `peek`, plus `passes-span` from the first pass begin to the last pass end and `between-passes` for copies and idle gaps). The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
The injector makes headset tests possible with nobody wearing the headset. The injector makes headset tests possible with nobody wearing the headset.
Keep the display awake, drive the menus, then take a compositor screenshot: Keep the display awake, drive the menus, then take a compositor screenshot:
@@ -649,6 +655,19 @@ at the start is about 1 ms of game-thread CPU per frame, with the GPU at 85 to
89%, so the next steps are on both sides: the guest-code share (translator 89%, so the next steps are on both sides: the guest-code share (translator
output quality) and the eye replay's GPU cost. output quality) and the eye replay's GPU cost.
The GPU side, measured the same day with per-pass timestamp queries (the second
`fpslog` line): on SNES Ghost Valley 2 at `render_scale` 0.5 (840x880 eyes) a
stereo frame cost 13.2 ms, of which the native render was 5.7 ms, the eyes 3.5
and 3.8, copies and gaps 0.4. That native render is a 1280x720 image nobody
sees during an immersive race, so it now stops after the last pass whose EFB
copy the eyes sample: `mono` fell to 0.15 ms and a Luigi Circuit start at 0.5
renders in 5.5 to 10 ms of GPU per frame. What remains is the headset pacing:
with the display at 72 or 90 Hz, each headset frame stays open for the next
60 Hz game frame plus the whole encode (`open` 16 ms in the pacing summary),
so cycles span one to two display slots and the headset gets 40 to 60 frames
per second while the game renders 60. Reworking that pacing (encode the newest
sealed frame at once, repeat the layer otherwise) is the next step.
Verified on device since: the menus on the virtual screen, controller input Verified on device since: the menus on the virtual screen, controller input
(the user has driven races), and an immersive Grand Prix start with all 12 (the user has driven races), and an immersive Grand Prix start with all 12
racers rendering correctly. Not yet verified: stereo comfort and scale, racers rendering correctly. Not yet verified: stereo comfort and scale,