Implement per-pass GPU timing for performance tracking and optimization & fixed mono in Immersive

This commit is contained in:
iChris4 committed 2026-09-19 19:23:57 +02:00
1 parent f0e5e43985
commit 47b59c7294
10 files changed
+344 -5

No files matched your search

+30 -2
View File
@@ -756,6 +756,7 @@ void encode_virtual_screen_eye(wgpu::CommandEncoder& encoder, const webgpu::Pres
.label = eyeIndex == 0 ? "Virtual screen left eye" : "Virtual screen right eye",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::VirtualScreen),
};
{
const auto pass = encoder.BeginRenderPass(&descriptor);
@@ -1274,6 +1275,7 @@ bool present_presentation_job(const PresentationJob& job) {
.label = "Presentation copy pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Present),
};
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(webgpu::g_CopyPipeline);
@@ -1576,6 +1578,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
.label = "Interpolation snapshot pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
};
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
const auto imageWidth = static_cast<float>(image.texture.size.width);
@@ -1627,6 +1630,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
.label = "Snapshot ImGui pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
};
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
pass.SetViewport(0.f, 0.f, static_cast<float>(image.texture.size.width),
@@ -1829,6 +1833,9 @@ void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept {
void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag,
const StereoSceneAnchor& sceneAnchor) {
ZoneScopedN("Seal frame");
// Every pass this cycle encodes, from the seal's probe blits to the final eye, is timed under
// one frame; encode_sealed_frame resolves it on its last submission.
gfx::gpu_timing_begin_frame();
const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{
.label = "Redraw encoder",
};
@@ -1988,7 +1995,18 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
// A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed
// stream would mutate an already-rendered EFB. Render once, then duplicate into the slots.
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo);
//
// On a headset an immersive frame's native render is never presented: the eyes replay the draws
// themselves and only sample the EFB copies it resolves. So it stops after the last pass that
// produces one of those copies (never the display copy), which on a Quest 3 was 4 to 6 ms of a
// 12 ms GPU frame spent on a 1280x720 image nobody saw. A pending CPU readback or a frame
// capture still gets the whole image.
int32_t nativeRenderLastPass = INT32_MAX;
if (headsetOnly && immersiveReplay && !gfx::efb_ram::has_pending() &&
g_captureFrame.load(std::memory_order_acquire) == UINT32_MAX) {
nativeRenderLastPass = gfx::last_pass_feeding_replay(sealedFrame);
}
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo, nativeRenderLastPass);
// The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder;
// completion is harvested in gfx::after_submit, never waited on here.
gfx::efb_ram::encode_async_downloads(encoder);
@@ -2070,7 +2088,9 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
.presentAt = slotPresentDeadline(ctx.interpolatedFrameCount),
.interpolated = false,
});
gfx::gpu_timing_end_frame(encoder);
submitEncodedSlot(encoder, pendingStereoSink ? &*pendingStereoSink : nullptr);
gfx::gpu_timing_after_submit();
// A group that finished encoding past its anchor slides forward by whole display periods, never
// per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period.
@@ -2180,7 +2200,12 @@ void record_frame_telemetry() {
{
// `adb shell setprop debug.wiicompiled.fpslog 1` before launch logs the game's rendered frame rate every five
// seconds. The headset compositor's own log (logcat tag VrApi) repeats frames, so it cannot show this.
static const bool fpsLog = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
static const bool fpsLog = [] {
const bool on = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
// The same switch turns on the per-pass GPU timestamps reported below the frame-rate line.
gfx::gpu_timing_set_enabled(on);
return on;
}();
if (fpsLog) {
static auto windowStart = std::chrono::steady_clock::now();
static uint32_t windowFrames = 0;
@@ -2203,6 +2228,9 @@ void record_frame_telemetry() {
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding",
windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal,
permitWait, prepare, encode);
if (const std::string gpuTiming = gfx::gpu_timing_report(); !gpuTiming.empty()) {
Log.info("{}", gpuTiming);
}
windowStart = now;
windowFrames = 0;
}
+230 -1
View File
@@ -1724,6 +1724,9 @@ struct RenderInvocation {
uint32_t localPlayerCount = 1;
// Inclusive index of the last pass to replay; -1 replays every pass.
int32_t replayLastPass = -1;
// Inclusive index of the last pass that does render work; texture bakes still run for the
// passes after it. See last_pass_feeding_replay.
int32_t renderLastPass = INT32_MAX;
bool finalize = true;
bool replayOnlyEfb = false;
bool skipCopyClears = false;
@@ -1755,6 +1758,11 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
tex_palette_conv::run(cmd, conv);
}
}
if (static_cast<int32_t>(i) > invocation.renderLastPass) {
// Nothing after the last replay-feeding resolve is shown or sampled on a headset; the
// bakes above are all these passes owe the eye replays.
continue;
}
const bool hasRenderWork = passInfo.clearColor || passInfo.clearDepth || !passInfo.commands.empty();
if (i == renderPasses.size() - 1) {
ASSERT(!passInfo.resolveTarget, "Final render pass must not have resolve target");
@@ -1793,11 +1801,16 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
.depthStoreOp = wgpu::StoreOp::Store,
.depthClearValue = passInfo.clearDepthValue,
};
const GpuTimingCategory timingCategory = invocation.stereoEye == 0 ? GpuTimingCategory::EyeLeft
: invocation.stereoEye == 1 ? GpuTimingCategory::EyeRight
: invocation.interpolatedFrame >= 0 ? GpuTimingCategory::Interpolated
: GpuTimingCategory::Mono;
const wgpu::RenderPassDescriptor renderPassDescriptor{
.label = render_pass_label(i),
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.depthStencilAttachment = &depthStencilAttachment,
.timestampWrites = gpu_timing_pass(timingCategory),
};
auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
@@ -1908,15 +1921,28 @@ void seal_frame(SealedFrame& out) noexcept {
g_currentRenderPass = UINT32_MAX;
}
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) {
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize,
int32_t nativeRenderLastPass) {
render_impl(frame.data().passes, cmd,
RenderInvocation{
.interpolatedFrame = interpolatedFrame,
.renderLastPass = nativeRenderLastPass,
.finalize = finalize,
.encodeTextureBakes = interpolatedFrame < 0,
});
}
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept {
const auto& passes = frame.data().passes;
int32_t last = -1;
for (size_t i = 0; i < passes.size(); ++i) {
if (passes[i].resolveTarget && !passes[i].displayCopyResolve) {
last = static_cast<int32_t>(i);
}
}
return last;
}
bool has_late_stereo_replay(const SealedFrame& frame) noexcept {
const auto& data = frame.data().stereo;
return data.generation != 0 && data.generation == g_replayBufferGeneration.load(std::memory_order_acquire) &&
@@ -2028,6 +2054,209 @@ void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize)
}
}
// --- Per-pass GPU timing (see common.hpp) -------------------------------------------------------
namespace {
constexpr uint32_t kGpuTimingSlots = 4;
constexpr uint32_t kGpuTimingPairs = 62;
constexpr uint32_t kGpuTimingQueries = 2 * kGpuTimingPairs;
struct GpuTimingSlot {
wgpu::QuerySet querySet;
wgpu::Buffer resolve;
wgpu::Buffer readback;
std::array<wgpu::PassTimestampWrites, kGpuTimingPairs> writes{};
std::array<GpuTimingCategory, kGpuTimingPairs> categories{};
uint32_t pairs = 0;
bool open = false; // between the frame's begin and end
bool reading = false; // readback in flight or mapped
bool mapped = false; // the callback ran; the encoding thread unmaps on reuse
};
std::atomic<bool> g_gpuTimingEnabled{false};
std::array<GpuTimingSlot, kGpuTimingSlots> g_gpuTimingSlots;
uint32_t g_gpuTimingNextSlot = 0;
int32_t g_gpuTimingCurrent = -1;
bool g_gpuTimingReady = false;
// Guards the totals below and every slot's reading/mapped flags: the map callback may run on
// whichever thread processes Dawn's events.
std::mutex g_gpuTimingMutex;
std::array<uint64_t, static_cast<size_t>(GpuTimingCategory::Count)> g_gpuTimingTotalsNs{};
uint64_t g_gpuTimingSpanNs = 0;
uint32_t g_gpuTimingFrames = 0;
uint32_t g_gpuTimingSkipped = 0;
bool gpu_timing_create_slots() {
if (g_gpuTimingReady) {
return true;
}
if (!webgpu::g_timestampQueriesSupported || !webgpu::g_device) {
return false;
}
for (auto& slot : g_gpuTimingSlots) {
const wgpu::QuerySetDescriptor querySetDescriptor{
.label = "GPU timing queries",
.type = wgpu::QueryType::Timestamp,
.count = kGpuTimingQueries,
};
slot.querySet = webgpu::g_device.CreateQuerySet(&querySetDescriptor);
const wgpu::BufferDescriptor resolveDescriptor{
.label = "GPU timing resolve",
.usage = wgpu::BufferUsage::QueryResolve | wgpu::BufferUsage::CopySrc,
.size = kGpuTimingQueries * sizeof(uint64_t),
};
slot.resolve = webgpu::g_device.CreateBuffer(&resolveDescriptor);
const wgpu::BufferDescriptor readbackDescriptor{
.label = "GPU timing readback",
.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst,
.size = kGpuTimingQueries * sizeof(uint64_t),
};
slot.readback = webgpu::g_device.CreateBuffer(&readbackDescriptor);
}
g_gpuTimingReady = true;
return true;
}
} // namespace
void gpu_timing_set_enabled(bool enabled) noexcept { g_gpuTimingEnabled.store(enabled, std::memory_order_relaxed); }
bool gpu_timing_enabled() noexcept { return g_gpuTimingEnabled.load(std::memory_order_relaxed); }
void gpu_timing_begin_frame() noexcept {
g_gpuTimingCurrent = -1;
if (!gpu_timing_enabled() || !gpu_timing_create_slots()) {
return;
}
const uint32_t index = g_gpuTimingNextSlot;
g_gpuTimingNextSlot = (g_gpuTimingNextSlot + 1) % kGpuTimingSlots;
auto& slot = g_gpuTimingSlots[index];
{
std::lock_guard lock(g_gpuTimingMutex);
if (slot.reading && !slot.mapped) {
++g_gpuTimingSkipped; // the GPU is more than a ring behind; leave this frame untimed
return;
}
if (slot.mapped) {
slot.readback.Unmap();
slot.mapped = false;
}
slot.reading = false;
}
slot.pairs = 0;
slot.open = true;
g_gpuTimingCurrent = static_cast<int32_t>(index);
}
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept {
if (g_gpuTimingCurrent < 0) {
return nullptr;
}
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
if (!slot.open || slot.pairs >= kGpuTimingPairs) {
return nullptr;
}
const uint32_t i = slot.pairs++;
slot.writes[i] = wgpu::PassTimestampWrites{
.querySet = slot.querySet,
.beginningOfPassWriteIndex = 2 * i,
.endOfPassWriteIndex = 2 * i + 1,
};
slot.categories[i] = category;
return &slot.writes[i];
}
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept {
if (g_gpuTimingCurrent < 0) {
return;
}
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
slot.open = false;
if (slot.pairs == 0) {
g_gpuTimingCurrent = -1;
return;
}
const uint32_t queries = 2 * slot.pairs;
encoder.ResolveQuerySet(slot.querySet, 0, queries, slot.resolve, 0);
encoder.CopyBufferToBuffer(slot.resolve, 0, slot.readback, 0, queries * sizeof(uint64_t));
}
void gpu_timing_after_submit() noexcept {
if (g_gpuTimingCurrent < 0) {
return;
}
const uint32_t index = static_cast<uint32_t>(g_gpuTimingCurrent);
g_gpuTimingCurrent = -1;
auto& slot = g_gpuTimingSlots[index];
const uint32_t pairs = slot.pairs;
{
std::lock_guard lock(g_gpuTimingMutex);
slot.reading = true;
slot.mapped = false;
}
slot.readback.MapAsync(
wgpu::MapMode::Read, 0, 2 * pairs * sizeof(uint64_t), wgpu::CallbackMode::AllowSpontaneous,
[index, pairs](wgpu::MapAsyncStatus status, wgpu::StringView) {
auto& slot = g_gpuTimingSlots[index];
std::lock_guard lock(g_gpuTimingMutex);
if (status != wgpu::MapAsyncStatus::Success) {
slot.reading = false;
return;
}
const auto* stamps =
static_cast<const uint64_t*>(slot.readback.GetConstMappedRange(0, 2 * pairs * sizeof(uint64_t)));
if (stamps != nullptr) {
uint64_t first = UINT64_MAX;
uint64_t last = 0;
for (uint32_t i = 0; i < pairs; ++i) {
const uint64_t begin = stamps[2 * i];
const uint64_t end = stamps[2 * i + 1];
if (end < begin) {
continue;
}
g_gpuTimingTotalsNs[static_cast<size_t>(slot.categories[i])] += end - begin;
first = std::min(first, begin);
last = std::max(last, end);
}
if (last > first) {
g_gpuTimingSpanNs += last - first;
}
++g_gpuTimingFrames;
}
slot.mapped = true;
});
}
std::string gpu_timing_report() {
std::lock_guard lock(g_gpuTimingMutex);
if (g_gpuTimingFrames == 0 && g_gpuTimingSkipped == 0) {
return {};
}
static constexpr std::array<const char*, static_cast<size_t>(GpuTimingCategory::Count)> kNames{
"mono", "eyeL", "eyeR", "interp", "screen", "panel", "efbcopy", "palette", "peek", "snapshot", "present"};
std::string text;
if (g_gpuTimingFrames != 0) {
const double frames = g_gpuTimingFrames;
uint64_t sum = 0;
text += fmt::format("GPU ms/frame over {} frames: passes-span={:.2f}", g_gpuTimingFrames,
static_cast<double>(g_gpuTimingSpanNs) / 1e6 / frames);
for (size_t i = 0; i < kNames.size(); ++i) {
if (g_gpuTimingTotalsNs[i] == 0) {
continue;
}
sum += g_gpuTimingTotalsNs[i];
text += fmt::format(" {}={:.2f}", kNames[i], static_cast<double>(g_gpuTimingTotalsNs[i]) / 1e6 / frames);
}
const uint64_t between = g_gpuTimingSpanNs > sum ? g_gpuTimingSpanNs - sum : 0;
text += fmt::format(" between-passes={:.2f}", static_cast<double>(between) / 1e6 / frames);
}
if (g_gpuTimingSkipped != 0) {
text += fmt::format(" (untimed frames: {})", g_gpuTimingSkipped);
}
g_gpuTimingTotalsNs.fill(0);
g_gpuTimingSpanNs = 0;
g_gpuTimingFrames = 0;
g_gpuTimingSkipped = 0;
return text;
}
void after_submit() noexcept {
depth_peek::after_submit();
efb_ram::after_submit();
+42 -1
View File
@@ -8,6 +8,7 @@
#include <cstring>
#include <array>
#include <memory>
#include <string>
#include <type_traits>
#include <utility>
@@ -358,7 +359,14 @@ bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, c
// Encode a sealed frame. Never touches the producer-visible recording state,
// so this may run concurrently with the producer's FIFO drains.
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true);
// `nativeRenderLastPass` limits the passes that do render work (texture bakes still run for every
// pass): a headset never shows an immersive frame's native render, so encode_sealed_frame stops it
// after the last pass whose EFB copy the eye replays sample.
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true,
int32_t nativeRenderLastPass = INT32_MAX);
// Index of the last recorded pass that resolves an EFB copy other than the display copy, or -1
// when no pass does: everything after it exists only for the presented image.
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept;
// Replays only main-EFB passes into one Aurora-owned eye target. Native
// offscreen/EFB-copy passes are consumed from the mono render and are not
@@ -414,6 +422,39 @@ bool is_offscreen() noexcept;
uint32_t get_sample_count() noexcept;
void clear_caches() noexcept;
// Per-pass GPU timing for the frame-rate log. When enabled and the device has TimestampQuery,
// every render or compute pass asks gpu_timing_pass() for timestamp writes under a category; the
// frame's queries are resolved into a small ring of readback buffers and the completed frames'
// durations are summed per category until gpu_timing_report() consumes them. Off by default:
// aurora.cpp enables it together with the Android frame-rate log.
enum class GpuTimingCategory : uint8_t {
Mono, // the native (desktop) render of the recorded GX passes
EyeLeft, // stereo replay of the left eye
EyeRight, // stereo replay of the right eye
Interpolated, // interpolated presentation slots
VirtualScreen, // the 2D virtual screen built for each eye
Panel, // the in-headset settings panel
EfbCopy, // EFB copy format conversions
Palette, // palette (TLUT) texture conversions
DepthPeek, // the depth snapshot compute pass
Snapshot, // presentation snapshot and its ImGui pass
Present, // the desktop presentation copy
Count,
};
void gpu_timing_set_enabled(bool enabled) noexcept;
bool gpu_timing_enabled() noexcept;
// Opens the current frame's query slot; a frame whose slot is still being read back is skipped.
void gpu_timing_begin_frame() noexcept;
// Timestamp writes for one pass of the open frame, or nullptr when timing is off or exhausted.
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept;
// Resolves the open frame's queries on `encoder`, which must be the frame's last submission.
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept;
// After that submission: starts the readback of the resolved queries.
void gpu_timing_after_submit() noexcept;
// Per-frame averages of the frames read back since the last call, formatted for the log, or
// an empty string when nothing was measured.
std::string gpu_timing_report();
namespace tex_palette_conv {
struct ConvRequest;
} // namespace tex_palette_conv
+2
View File
@@ -1,4 +1,5 @@
#include "depth_peek.hpp"
#include "common.hpp"
#include "../dolphin/vi/vi_internal.hpp"
#include "../gx/gx.hpp"
@@ -403,6 +404,7 @@ void encode_frame_snapshot(const wgpu::CommandEncoder& cmd, const wgpu::TextureV
const wgpu::ComputePassDescriptor passDescriptor{
.label = "Depth Peek Compute Pass",
.timestampWrites = gpu_timing_pass(GpuTimingCategory::DepthPeek),
};
const auto pass = cmd.BeginComputePass(&passDescriptor);
pass.SetPipeline(g_pipeline);
+2
View File
@@ -1,4 +1,5 @@
#include "tex_copy_conv.hpp"
#include "common.hpp"
#include "tex_copy_format_contract.hpp"
#include "../internal.hpp"
@@ -586,6 +587,7 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con
.label = "TexCopyConv Pass",
.colorAttachmentCount = colorAttachments.size(),
.colorAttachments = colorAttachments.data(),
.timestampWrites = gpu_timing_pass(GpuTimingCategory::EfbCopy),
};
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(pipeline);
+2
View File
@@ -1,4 +1,5 @@
#include "tex_palette_conv.hpp"
#include "common.hpp"
#include "../internal.hpp"
#include "../webgpu/gpu.hpp"
@@ -249,6 +250,7 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
.label = "TexPaletteConv Pass",
.colorAttachmentCount = colorAttachments.size(),
.colorAttachments = colorAttachments.data(),
.timestampWrites = gpu_timing_pass(GpuTimingCategory::Palette),
};
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(pipeline);
+2
View File
@@ -239,6 +239,7 @@ void composite(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& tar
.label = eyeIndex == 0 ? "Headset panel left eye" : "Headset panel right eye",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
};
const auto pass = encoder.BeginRenderPass(&descriptor);
pass.SetPipeline(state.pipeline);
@@ -294,6 +295,7 @@ wgpu::CommandBuffer prepare(ImDrawData* drawData, float widthFraction) noexcept
.label = "Headset panel ImGui pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
};
bool drawn = false;
{
+12
View File
@@ -80,6 +80,7 @@ wgpu::Instance g_instance;
static wgpu::AdapterInfo g_adapterInfo;
static wgpu::SurfaceCapabilities g_surfaceCapabilities;
bool g_bcTexturesSupported;
bool g_timestampQueriesSupported = false;
// Written by Dawn's device-loss callback and consumed at ordered frame boundaries. Keep the
// callback free of logging, allocation, teardown and renderer state mutation.
static std::atomic_bool g_deviceLost{false};
@@ -702,6 +703,12 @@ bool initialize(AuroraBackend auroraBackend) {
g_bcTexturesSupported = true;
requiredFeatures.push_back(feature);
}
// Per-pass GPU timing for the frame-rate log (gfx::gpu_timing_*). Requesting the feature
// costs nothing until a pass carries timestamp writes.
if (feature == wgpu::FeatureName::TimestampQuery) {
g_timestampQueriesSupported = true;
requiredFeatures.push_back(feature);
}
// The presenter calls device and queue methods while the frame worker encodes, which Dawn only
// supports with this feature; without it the two race inside the device's dynamic uploader.
if (feature == wgpu::FeatureName::ImplicitDeviceSynchronization) {
@@ -769,10 +776,15 @@ bool initialize(AuroraBackend auroraBackend) {
if (g_backendType == wgpu::BackendType::Vulkan) {
enableToggles.push_back("vulkan_monolithic_pipeline_cache");
}
// Dawn quantizes timestamp queries to 100 us for web privacy; the per-pass GPU timing wants
// the raw values.
const std::array<const char*, 1> disableToggles{"timestamp_quantization"};
const wgpu::DawnTogglesDescriptor togglesDescriptor({
.nextInChain = &cacheDescriptor,
.enabledToggleCount = enableToggles.size(),
.enabledToggles = enableToggles.data(),
.disabledToggleCount = g_timestampQueriesSupported ? disableToggles.size() : 0,
.disabledToggles = disableToggles.data(),
});
#endif
wgpu::DeviceDescriptor deviceDescriptor;
+2
View File
@@ -58,6 +58,8 @@ extern wgpu::RenderPipeline g_CopyPipeline;
extern wgpu::BindGroup g_CopyBindGroup;
extern wgpu::Instance g_instance;
extern bool g_bcTexturesSupported;
// The device was created with TimestampQuery, so passes may carry timestamp writes (gfx::gpu_timing_*).
extern bool g_timestampQueriesSupported;
bool initialize(AuroraBackend backend);
void shutdown();
+20 -1
View File
@@ -143,6 +143,12 @@ suggested for `oculus/touch_controller` and `khr/simple_controller`.
provider is registered on Android, Aurora skips the surface present and the
desktop mirror copy (`headset_owns_display` in `lib/aurora.cpp`). The game's
own render size is unaffected: at `resolution_multiplier = 1` it is 640x528.
Since 2026-09-19 an immersive race also stops that native render after the
last pass whose EFB copy the eye replays sample (`last_pass_feeding_replay`):
the main scene and display copy of a 1280x720 image nobody sees were 4 to
6 ms of a 12 ms GPU frame on a Quest 3. A pending CPU readback of an EFB
copy or a frame capture still renders the whole image, and menus (the
virtual screen) keep it because their eyes are built from that snapshot.
- **JNI only on the real thread stack.** Guest threads run on libco stacks
inside the SDL thread, and SDL's Android event pump can reach Java (joystick
polling, HIDAPI). ART binds JNI transitions to the thread's real stack, so
@@ -582,7 +588,7 @@ the app:
| `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update |
| `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds |
| `debug.wiicompiled.inject <n>:<button>` | Presses `a`, `b`, `x`, `y`, `start`, `up`, `down`, `left` or `right` for 12 XR frames each time `<n>` changes. As a Wii Remote, `x`/`y`/`start` are 1/2/+, the directions push the Nunchuk stick, and `home`, `c` and `z` also exist. `panel` presses the settings panel's button (left Y, or both thumbsticks as a gamepad), opening or closing it (see `OPENXR.md`) |
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. A second line gives the GPU time per frame from timestamp queries on every pass (`mono` native render, `eyeL`/`eyeR` replays, `screen`, `panel`, `efbcopy`, `palette`, `peek`, plus `passes-span` from the first pass begin to the last pass end and `between-passes` for copies and idle gaps). The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
The injector makes headset tests possible with nobody wearing the headset.
Keep the display awake, drive the menus, then take a compositor screenshot:
@@ -649,6 +655,19 @@ at the start is about 1 ms of game-thread CPU per frame, with the GPU at 85 to
89%, so the next steps are on both sides: the guest-code share (translator
output quality) and the eye replay's GPU cost.
The GPU side, measured the same day with per-pass timestamp queries (the second
`fpslog` line): on SNES Ghost Valley 2 at `render_scale` 0.5 (840x880 eyes) a
stereo frame cost 13.2 ms, of which the native render was 5.7 ms, the eyes 3.5
and 3.8, copies and gaps 0.4. That native render is a 1280x720 image nobody
sees during an immersive race, so it now stops after the last pass whose EFB
copy the eyes sample: `mono` fell to 0.15 ms and a Luigi Circuit start at 0.5
renders in 5.5 to 10 ms of GPU per frame. What remains is the headset pacing:
with the display at 72 or 90 Hz, each headset frame stays open for the next
60 Hz game frame plus the whole encode (`open` 16 ms in the pacing summary),
so cycles span one to two display slots and the headset gets 40 to 60 frames
per second while the game renders 60. Reworking that pacing (encode the newest
sealed frame at once, repeat the layer otherwise) is the next step.
Verified on device since: the menus on the virtual screen, controller input
(the user has driven races), and an immersive Grand Prix start with all 12
racers rendering correctly. Not yet verified: stereo comfort and scale,