mirror of
https://github.com/mitch030504/Wiicompiled_VR_Frame.git
synced 2026-10-06 04:04:18 +02:00
Implement per-pass GPU timing for performance tracking and optimization & fixed mono in Immersive
This commit is contained in:
1 parent
f0e5e43985
commit
47b59c7294
10 files changed
+344
-5
No files matched your search
@@ -756,6 +756,7 @@ void encode_virtual_screen_eye(wgpu::CommandEncoder& encoder, const webgpu::Pres
|
||||
.label = eyeIndex == 0 ? "Virtual screen left eye" : "Virtual screen right eye",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::VirtualScreen),
|
||||
};
|
||||
{
|
||||
const auto pass = encoder.BeginRenderPass(&descriptor);
|
||||
@@ -1274,6 +1275,7 @@ bool present_presentation_job(const PresentationJob& job) {
|
||||
.label = "Presentation copy pass",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Present),
|
||||
};
|
||||
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
|
||||
pass.SetPipeline(webgpu::g_CopyPipeline);
|
||||
@@ -1576,6 +1578,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
|
||||
.label = "Interpolation snapshot pass",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
|
||||
};
|
||||
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
|
||||
const auto imageWidth = static_cast<float>(image.texture.size.width);
|
||||
@@ -1627,6 +1630,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
|
||||
.label = "Snapshot ImGui pass",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
|
||||
};
|
||||
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
|
||||
pass.SetViewport(0.f, 0.f, static_cast<float>(image.texture.size.width),
|
||||
@@ -1829,6 +1833,9 @@ void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept {
|
||||
void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag,
|
||||
const StereoSceneAnchor& sceneAnchor) {
|
||||
ZoneScopedN("Seal frame");
|
||||
// Every pass this cycle encodes, from the seal's probe blits to the final eye, is timed under
|
||||
// one frame; encode_sealed_frame resolves it on its last submission.
|
||||
gfx::gpu_timing_begin_frame();
|
||||
const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{
|
||||
.label = "Redraw encoder",
|
||||
};
|
||||
@@ -1988,7 +1995,18 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
|
||||
// A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed
|
||||
// stream would mutate an already-rendered EFB. Render once, then duplicate into the slots.
|
||||
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo);
|
||||
//
|
||||
// On a headset an immersive frame's native render is never presented: the eyes replay the draws
|
||||
// themselves and only sample the EFB copies it resolves. So it stops after the last pass that
|
||||
// produces one of those copies (never the display copy), which on a Quest 3 was 4 to 6 ms of a
|
||||
// 12 ms GPU frame spent on a 1280x720 image nobody saw. A pending CPU readback or a frame
|
||||
// capture still gets the whole image.
|
||||
int32_t nativeRenderLastPass = INT32_MAX;
|
||||
if (headsetOnly && immersiveReplay && !gfx::efb_ram::has_pending() &&
|
||||
g_captureFrame.load(std::memory_order_acquire) == UINT32_MAX) {
|
||||
nativeRenderLastPass = gfx::last_pass_feeding_replay(sealedFrame);
|
||||
}
|
||||
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo, nativeRenderLastPass);
|
||||
// The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder;
|
||||
// completion is harvested in gfx::after_submit, never waited on here.
|
||||
gfx::efb_ram::encode_async_downloads(encoder);
|
||||
@@ -2070,7 +2088,9 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
.presentAt = slotPresentDeadline(ctx.interpolatedFrameCount),
|
||||
.interpolated = false,
|
||||
});
|
||||
gfx::gpu_timing_end_frame(encoder);
|
||||
submitEncodedSlot(encoder, pendingStereoSink ? &*pendingStereoSink : nullptr);
|
||||
gfx::gpu_timing_after_submit();
|
||||
|
||||
// A group that finished encoding past its anchor slides forward by whole display periods, never
|
||||
// per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period.
|
||||
@@ -2180,7 +2200,12 @@ void record_frame_telemetry() {
|
||||
{
|
||||
// `adb shell setprop debug.wiicompiled.fpslog 1` before launch logs the game's rendered frame rate every five
|
||||
// seconds. The headset compositor's own log (logcat tag VrApi) repeats frames, so it cannot show this.
|
||||
static const bool fpsLog = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
|
||||
static const bool fpsLog = [] {
|
||||
const bool on = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
|
||||
// The same switch turns on the per-pass GPU timestamps reported below the frame-rate line.
|
||||
gfx::gpu_timing_set_enabled(on);
|
||||
return on;
|
||||
}();
|
||||
if (fpsLog) {
|
||||
static auto windowStart = std::chrono::steady_clock::now();
|
||||
static uint32_t windowFrames = 0;
|
||||
@@ -2203,6 +2228,9 @@ void record_frame_telemetry() {
|
||||
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding",
|
||||
windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal,
|
||||
permitWait, prepare, encode);
|
||||
if (const std::string gpuTiming = gfx::gpu_timing_report(); !gpuTiming.empty()) {
|
||||
Log.info("{}", gpuTiming);
|
||||
}
|
||||
windowStart = now;
|
||||
windowFrames = 0;
|
||||
}
|
||||
|
||||
@@ -1724,6 +1724,9 @@ struct RenderInvocation {
|
||||
uint32_t localPlayerCount = 1;
|
||||
// Inclusive index of the last pass to replay; -1 replays every pass.
|
||||
int32_t replayLastPass = -1;
|
||||
// Inclusive index of the last pass that does render work; texture bakes still run for the
|
||||
// passes after it. See last_pass_feeding_replay.
|
||||
int32_t renderLastPass = INT32_MAX;
|
||||
bool finalize = true;
|
||||
bool replayOnlyEfb = false;
|
||||
bool skipCopyClears = false;
|
||||
@@ -1755,6 +1758,11 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
|
||||
tex_palette_conv::run(cmd, conv);
|
||||
}
|
||||
}
|
||||
if (static_cast<int32_t>(i) > invocation.renderLastPass) {
|
||||
// Nothing after the last replay-feeding resolve is shown or sampled on a headset; the
|
||||
// bakes above are all these passes owe the eye replays.
|
||||
continue;
|
||||
}
|
||||
const bool hasRenderWork = passInfo.clearColor || passInfo.clearDepth || !passInfo.commands.empty();
|
||||
if (i == renderPasses.size() - 1) {
|
||||
ASSERT(!passInfo.resolveTarget, "Final render pass must not have resolve target");
|
||||
@@ -1793,11 +1801,16 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
|
||||
.depthStoreOp = wgpu::StoreOp::Store,
|
||||
.depthClearValue = passInfo.clearDepthValue,
|
||||
};
|
||||
const GpuTimingCategory timingCategory = invocation.stereoEye == 0 ? GpuTimingCategory::EyeLeft
|
||||
: invocation.stereoEye == 1 ? GpuTimingCategory::EyeRight
|
||||
: invocation.interpolatedFrame >= 0 ? GpuTimingCategory::Interpolated
|
||||
: GpuTimingCategory::Mono;
|
||||
const wgpu::RenderPassDescriptor renderPassDescriptor{
|
||||
.label = render_pass_label(i),
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.depthStencilAttachment = &depthStencilAttachment,
|
||||
.timestampWrites = gpu_timing_pass(timingCategory),
|
||||
};
|
||||
|
||||
auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
|
||||
@@ -1908,15 +1921,28 @@ void seal_frame(SealedFrame& out) noexcept {
|
||||
g_currentRenderPass = UINT32_MAX;
|
||||
}
|
||||
|
||||
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) {
|
||||
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize,
|
||||
int32_t nativeRenderLastPass) {
|
||||
render_impl(frame.data().passes, cmd,
|
||||
RenderInvocation{
|
||||
.interpolatedFrame = interpolatedFrame,
|
||||
.renderLastPass = nativeRenderLastPass,
|
||||
.finalize = finalize,
|
||||
.encodeTextureBakes = interpolatedFrame < 0,
|
||||
});
|
||||
}
|
||||
|
||||
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept {
|
||||
const auto& passes = frame.data().passes;
|
||||
int32_t last = -1;
|
||||
for (size_t i = 0; i < passes.size(); ++i) {
|
||||
if (passes[i].resolveTarget && !passes[i].displayCopyResolve) {
|
||||
last = static_cast<int32_t>(i);
|
||||
}
|
||||
}
|
||||
return last;
|
||||
}
|
||||
|
||||
bool has_late_stereo_replay(const SealedFrame& frame) noexcept {
|
||||
const auto& data = frame.data().stereo;
|
||||
return data.generation != 0 && data.generation == g_replayBufferGeneration.load(std::memory_order_acquire) &&
|
||||
@@ -2028,6 +2054,209 @@ void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Per-pass GPU timing (see common.hpp) -------------------------------------------------------
|
||||
namespace {
|
||||
constexpr uint32_t kGpuTimingSlots = 4;
|
||||
constexpr uint32_t kGpuTimingPairs = 62;
|
||||
constexpr uint32_t kGpuTimingQueries = 2 * kGpuTimingPairs;
|
||||
|
||||
struct GpuTimingSlot {
|
||||
wgpu::QuerySet querySet;
|
||||
wgpu::Buffer resolve;
|
||||
wgpu::Buffer readback;
|
||||
std::array<wgpu::PassTimestampWrites, kGpuTimingPairs> writes{};
|
||||
std::array<GpuTimingCategory, kGpuTimingPairs> categories{};
|
||||
uint32_t pairs = 0;
|
||||
bool open = false; // between the frame's begin and end
|
||||
bool reading = false; // readback in flight or mapped
|
||||
bool mapped = false; // the callback ran; the encoding thread unmaps on reuse
|
||||
};
|
||||
|
||||
std::atomic<bool> g_gpuTimingEnabled{false};
|
||||
std::array<GpuTimingSlot, kGpuTimingSlots> g_gpuTimingSlots;
|
||||
uint32_t g_gpuTimingNextSlot = 0;
|
||||
int32_t g_gpuTimingCurrent = -1;
|
||||
bool g_gpuTimingReady = false;
|
||||
// Guards the totals below and every slot's reading/mapped flags: the map callback may run on
|
||||
// whichever thread processes Dawn's events.
|
||||
std::mutex g_gpuTimingMutex;
|
||||
std::array<uint64_t, static_cast<size_t>(GpuTimingCategory::Count)> g_gpuTimingTotalsNs{};
|
||||
uint64_t g_gpuTimingSpanNs = 0;
|
||||
uint32_t g_gpuTimingFrames = 0;
|
||||
uint32_t g_gpuTimingSkipped = 0;
|
||||
|
||||
bool gpu_timing_create_slots() {
|
||||
if (g_gpuTimingReady) {
|
||||
return true;
|
||||
}
|
||||
if (!webgpu::g_timestampQueriesSupported || !webgpu::g_device) {
|
||||
return false;
|
||||
}
|
||||
for (auto& slot : g_gpuTimingSlots) {
|
||||
const wgpu::QuerySetDescriptor querySetDescriptor{
|
||||
.label = "GPU timing queries",
|
||||
.type = wgpu::QueryType::Timestamp,
|
||||
.count = kGpuTimingQueries,
|
||||
};
|
||||
slot.querySet = webgpu::g_device.CreateQuerySet(&querySetDescriptor);
|
||||
const wgpu::BufferDescriptor resolveDescriptor{
|
||||
.label = "GPU timing resolve",
|
||||
.usage = wgpu::BufferUsage::QueryResolve | wgpu::BufferUsage::CopySrc,
|
||||
.size = kGpuTimingQueries * sizeof(uint64_t),
|
||||
};
|
||||
slot.resolve = webgpu::g_device.CreateBuffer(&resolveDescriptor);
|
||||
const wgpu::BufferDescriptor readbackDescriptor{
|
||||
.label = "GPU timing readback",
|
||||
.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst,
|
||||
.size = kGpuTimingQueries * sizeof(uint64_t),
|
||||
};
|
||||
slot.readback = webgpu::g_device.CreateBuffer(&readbackDescriptor);
|
||||
}
|
||||
g_gpuTimingReady = true;
|
||||
return true;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void gpu_timing_set_enabled(bool enabled) noexcept { g_gpuTimingEnabled.store(enabled, std::memory_order_relaxed); }
|
||||
bool gpu_timing_enabled() noexcept { return g_gpuTimingEnabled.load(std::memory_order_relaxed); }
|
||||
|
||||
void gpu_timing_begin_frame() noexcept {
|
||||
g_gpuTimingCurrent = -1;
|
||||
if (!gpu_timing_enabled() || !gpu_timing_create_slots()) {
|
||||
return;
|
||||
}
|
||||
const uint32_t index = g_gpuTimingNextSlot;
|
||||
g_gpuTimingNextSlot = (g_gpuTimingNextSlot + 1) % kGpuTimingSlots;
|
||||
auto& slot = g_gpuTimingSlots[index];
|
||||
{
|
||||
std::lock_guard lock(g_gpuTimingMutex);
|
||||
if (slot.reading && !slot.mapped) {
|
||||
++g_gpuTimingSkipped; // the GPU is more than a ring behind; leave this frame untimed
|
||||
return;
|
||||
}
|
||||
if (slot.mapped) {
|
||||
slot.readback.Unmap();
|
||||
slot.mapped = false;
|
||||
}
|
||||
slot.reading = false;
|
||||
}
|
||||
slot.pairs = 0;
|
||||
slot.open = true;
|
||||
g_gpuTimingCurrent = static_cast<int32_t>(index);
|
||||
}
|
||||
|
||||
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept {
|
||||
if (g_gpuTimingCurrent < 0) {
|
||||
return nullptr;
|
||||
}
|
||||
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
|
||||
if (!slot.open || slot.pairs >= kGpuTimingPairs) {
|
||||
return nullptr;
|
||||
}
|
||||
const uint32_t i = slot.pairs++;
|
||||
slot.writes[i] = wgpu::PassTimestampWrites{
|
||||
.querySet = slot.querySet,
|
||||
.beginningOfPassWriteIndex = 2 * i,
|
||||
.endOfPassWriteIndex = 2 * i + 1,
|
||||
};
|
||||
slot.categories[i] = category;
|
||||
return &slot.writes[i];
|
||||
}
|
||||
|
||||
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept {
|
||||
if (g_gpuTimingCurrent < 0) {
|
||||
return;
|
||||
}
|
||||
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
|
||||
slot.open = false;
|
||||
if (slot.pairs == 0) {
|
||||
g_gpuTimingCurrent = -1;
|
||||
return;
|
||||
}
|
||||
const uint32_t queries = 2 * slot.pairs;
|
||||
encoder.ResolveQuerySet(slot.querySet, 0, queries, slot.resolve, 0);
|
||||
encoder.CopyBufferToBuffer(slot.resolve, 0, slot.readback, 0, queries * sizeof(uint64_t));
|
||||
}
|
||||
|
||||
void gpu_timing_after_submit() noexcept {
|
||||
if (g_gpuTimingCurrent < 0) {
|
||||
return;
|
||||
}
|
||||
const uint32_t index = static_cast<uint32_t>(g_gpuTimingCurrent);
|
||||
g_gpuTimingCurrent = -1;
|
||||
auto& slot = g_gpuTimingSlots[index];
|
||||
const uint32_t pairs = slot.pairs;
|
||||
{
|
||||
std::lock_guard lock(g_gpuTimingMutex);
|
||||
slot.reading = true;
|
||||
slot.mapped = false;
|
||||
}
|
||||
slot.readback.MapAsync(
|
||||
wgpu::MapMode::Read, 0, 2 * pairs * sizeof(uint64_t), wgpu::CallbackMode::AllowSpontaneous,
|
||||
[index, pairs](wgpu::MapAsyncStatus status, wgpu::StringView) {
|
||||
auto& slot = g_gpuTimingSlots[index];
|
||||
std::lock_guard lock(g_gpuTimingMutex);
|
||||
if (status != wgpu::MapAsyncStatus::Success) {
|
||||
slot.reading = false;
|
||||
return;
|
||||
}
|
||||
const auto* stamps =
|
||||
static_cast<const uint64_t*>(slot.readback.GetConstMappedRange(0, 2 * pairs * sizeof(uint64_t)));
|
||||
if (stamps != nullptr) {
|
||||
uint64_t first = UINT64_MAX;
|
||||
uint64_t last = 0;
|
||||
for (uint32_t i = 0; i < pairs; ++i) {
|
||||
const uint64_t begin = stamps[2 * i];
|
||||
const uint64_t end = stamps[2 * i + 1];
|
||||
if (end < begin) {
|
||||
continue;
|
||||
}
|
||||
g_gpuTimingTotalsNs[static_cast<size_t>(slot.categories[i])] += end - begin;
|
||||
first = std::min(first, begin);
|
||||
last = std::max(last, end);
|
||||
}
|
||||
if (last > first) {
|
||||
g_gpuTimingSpanNs += last - first;
|
||||
}
|
||||
++g_gpuTimingFrames;
|
||||
}
|
||||
slot.mapped = true;
|
||||
});
|
||||
}
|
||||
|
||||
std::string gpu_timing_report() {
|
||||
std::lock_guard lock(g_gpuTimingMutex);
|
||||
if (g_gpuTimingFrames == 0 && g_gpuTimingSkipped == 0) {
|
||||
return {};
|
||||
}
|
||||
static constexpr std::array<const char*, static_cast<size_t>(GpuTimingCategory::Count)> kNames{
|
||||
"mono", "eyeL", "eyeR", "interp", "screen", "panel", "efbcopy", "palette", "peek", "snapshot", "present"};
|
||||
std::string text;
|
||||
if (g_gpuTimingFrames != 0) {
|
||||
const double frames = g_gpuTimingFrames;
|
||||
uint64_t sum = 0;
|
||||
text += fmt::format("GPU ms/frame over {} frames: passes-span={:.2f}", g_gpuTimingFrames,
|
||||
static_cast<double>(g_gpuTimingSpanNs) / 1e6 / frames);
|
||||
for (size_t i = 0; i < kNames.size(); ++i) {
|
||||
if (g_gpuTimingTotalsNs[i] == 0) {
|
||||
continue;
|
||||
}
|
||||
sum += g_gpuTimingTotalsNs[i];
|
||||
text += fmt::format(" {}={:.2f}", kNames[i], static_cast<double>(g_gpuTimingTotalsNs[i]) / 1e6 / frames);
|
||||
}
|
||||
const uint64_t between = g_gpuTimingSpanNs > sum ? g_gpuTimingSpanNs - sum : 0;
|
||||
text += fmt::format(" between-passes={:.2f}", static_cast<double>(between) / 1e6 / frames);
|
||||
}
|
||||
if (g_gpuTimingSkipped != 0) {
|
||||
text += fmt::format(" (untimed frames: {})", g_gpuTimingSkipped);
|
||||
}
|
||||
g_gpuTimingTotalsNs.fill(0);
|
||||
g_gpuTimingSpanNs = 0;
|
||||
g_gpuTimingFrames = 0;
|
||||
g_gpuTimingSkipped = 0;
|
||||
return text;
|
||||
}
|
||||
|
||||
void after_submit() noexcept {
|
||||
depth_peek::after_submit();
|
||||
efb_ram::after_submit();
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <cstring>
|
||||
#include <array>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
@@ -358,7 +359,14 @@ bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, c
|
||||
|
||||
// Encode a sealed frame. Never touches the producer-visible recording state,
|
||||
// so this may run concurrently with the producer's FIFO drains.
|
||||
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true);
|
||||
// `nativeRenderLastPass` limits the passes that do render work (texture bakes still run for every
|
||||
// pass): a headset never shows an immersive frame's native render, so encode_sealed_frame stops it
|
||||
// after the last pass whose EFB copy the eye replays sample.
|
||||
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true,
|
||||
int32_t nativeRenderLastPass = INT32_MAX);
|
||||
// Index of the last recorded pass that resolves an EFB copy other than the display copy, or -1
|
||||
// when no pass does: everything after it exists only for the presented image.
|
||||
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept;
|
||||
|
||||
// Replays only main-EFB passes into one Aurora-owned eye target. Native
|
||||
// offscreen/EFB-copy passes are consumed from the mono render and are not
|
||||
@@ -414,6 +422,39 @@ bool is_offscreen() noexcept;
|
||||
uint32_t get_sample_count() noexcept;
|
||||
void clear_caches() noexcept;
|
||||
|
||||
// Per-pass GPU timing for the frame-rate log. When enabled and the device has TimestampQuery,
|
||||
// every render or compute pass asks gpu_timing_pass() for timestamp writes under a category; the
|
||||
// frame's queries are resolved into a small ring of readback buffers and the completed frames'
|
||||
// durations are summed per category until gpu_timing_report() consumes them. Off by default:
|
||||
// aurora.cpp enables it together with the Android frame-rate log.
|
||||
enum class GpuTimingCategory : uint8_t {
|
||||
Mono, // the native (desktop) render of the recorded GX passes
|
||||
EyeLeft, // stereo replay of the left eye
|
||||
EyeRight, // stereo replay of the right eye
|
||||
Interpolated, // interpolated presentation slots
|
||||
VirtualScreen, // the 2D virtual screen built for each eye
|
||||
Panel, // the in-headset settings panel
|
||||
EfbCopy, // EFB copy format conversions
|
||||
Palette, // palette (TLUT) texture conversions
|
||||
DepthPeek, // the depth snapshot compute pass
|
||||
Snapshot, // presentation snapshot and its ImGui pass
|
||||
Present, // the desktop presentation copy
|
||||
Count,
|
||||
};
|
||||
void gpu_timing_set_enabled(bool enabled) noexcept;
|
||||
bool gpu_timing_enabled() noexcept;
|
||||
// Opens the current frame's query slot; a frame whose slot is still being read back is skipped.
|
||||
void gpu_timing_begin_frame() noexcept;
|
||||
// Timestamp writes for one pass of the open frame, or nullptr when timing is off or exhausted.
|
||||
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept;
|
||||
// Resolves the open frame's queries on `encoder`, which must be the frame's last submission.
|
||||
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept;
|
||||
// After that submission: starts the readback of the resolved queries.
|
||||
void gpu_timing_after_submit() noexcept;
|
||||
// Per-frame averages of the frames read back since the last call, formatted for the log, or
|
||||
// an empty string when nothing was measured.
|
||||
std::string gpu_timing_report();
|
||||
|
||||
namespace tex_palette_conv {
|
||||
struct ConvRequest;
|
||||
} // namespace tex_palette_conv
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "depth_peek.hpp"
|
||||
#include "common.hpp"
|
||||
|
||||
#include "../dolphin/vi/vi_internal.hpp"
|
||||
#include "../gx/gx.hpp"
|
||||
@@ -403,6 +404,7 @@ void encode_frame_snapshot(const wgpu::CommandEncoder& cmd, const wgpu::TextureV
|
||||
|
||||
const wgpu::ComputePassDescriptor passDescriptor{
|
||||
.label = "Depth Peek Compute Pass",
|
||||
.timestampWrites = gpu_timing_pass(GpuTimingCategory::DepthPeek),
|
||||
};
|
||||
const auto pass = cmd.BeginComputePass(&passDescriptor);
|
||||
pass.SetPipeline(g_pipeline);
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "tex_copy_conv.hpp"
|
||||
#include "common.hpp"
|
||||
#include "tex_copy_format_contract.hpp"
|
||||
|
||||
#include "../internal.hpp"
|
||||
@@ -586,6 +587,7 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con
|
||||
.label = "TexCopyConv Pass",
|
||||
.colorAttachmentCount = colorAttachments.size(),
|
||||
.colorAttachments = colorAttachments.data(),
|
||||
.timestampWrites = gpu_timing_pass(GpuTimingCategory::EfbCopy),
|
||||
};
|
||||
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
|
||||
pass.SetPipeline(pipeline);
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "tex_palette_conv.hpp"
|
||||
#include "common.hpp"
|
||||
|
||||
#include "../internal.hpp"
|
||||
#include "../webgpu/gpu.hpp"
|
||||
@@ -249,6 +250,7 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
|
||||
.label = "TexPaletteConv Pass",
|
||||
.colorAttachmentCount = colorAttachments.size(),
|
||||
.colorAttachments = colorAttachments.data(),
|
||||
.timestampWrites = gpu_timing_pass(GpuTimingCategory::Palette),
|
||||
};
|
||||
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
|
||||
pass.SetPipeline(pipeline);
|
||||
|
||||
@@ -239,6 +239,7 @@ void composite(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& tar
|
||||
.label = eyeIndex == 0 ? "Headset panel left eye" : "Headset panel right eye",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
|
||||
};
|
||||
const auto pass = encoder.BeginRenderPass(&descriptor);
|
||||
pass.SetPipeline(state.pipeline);
|
||||
@@ -294,6 +295,7 @@ wgpu::CommandBuffer prepare(ImDrawData* drawData, float widthFraction) noexcept
|
||||
.label = "Headset panel ImGui pass",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
|
||||
};
|
||||
bool drawn = false;
|
||||
{
|
||||
|
||||
@@ -80,6 +80,7 @@ wgpu::Instance g_instance;
|
||||
static wgpu::AdapterInfo g_adapterInfo;
|
||||
static wgpu::SurfaceCapabilities g_surfaceCapabilities;
|
||||
bool g_bcTexturesSupported;
|
||||
bool g_timestampQueriesSupported = false;
|
||||
// Written by Dawn's device-loss callback and consumed at ordered frame boundaries. Keep the
|
||||
// callback free of logging, allocation, teardown and renderer state mutation.
|
||||
static std::atomic_bool g_deviceLost{false};
|
||||
@@ -702,6 +703,12 @@ bool initialize(AuroraBackend auroraBackend) {
|
||||
g_bcTexturesSupported = true;
|
||||
requiredFeatures.push_back(feature);
|
||||
}
|
||||
// Per-pass GPU timing for the frame-rate log (gfx::gpu_timing_*). Requesting the feature
|
||||
// costs nothing until a pass carries timestamp writes.
|
||||
if (feature == wgpu::FeatureName::TimestampQuery) {
|
||||
g_timestampQueriesSupported = true;
|
||||
requiredFeatures.push_back(feature);
|
||||
}
|
||||
// The presenter calls device and queue methods while the frame worker encodes, which Dawn only
|
||||
// supports with this feature; without it the two race inside the device's dynamic uploader.
|
||||
if (feature == wgpu::FeatureName::ImplicitDeviceSynchronization) {
|
||||
@@ -769,10 +776,15 @@ bool initialize(AuroraBackend auroraBackend) {
|
||||
if (g_backendType == wgpu::BackendType::Vulkan) {
|
||||
enableToggles.push_back("vulkan_monolithic_pipeline_cache");
|
||||
}
|
||||
// Dawn quantizes timestamp queries to 100 us for web privacy; the per-pass GPU timing wants
|
||||
// the raw values.
|
||||
const std::array<const char*, 1> disableToggles{"timestamp_quantization"};
|
||||
const wgpu::DawnTogglesDescriptor togglesDescriptor({
|
||||
.nextInChain = &cacheDescriptor,
|
||||
.enabledToggleCount = enableToggles.size(),
|
||||
.enabledToggles = enableToggles.data(),
|
||||
.disabledToggleCount = g_timestampQueriesSupported ? disableToggles.size() : 0,
|
||||
.disabledToggles = disableToggles.data(),
|
||||
});
|
||||
#endif
|
||||
wgpu::DeviceDescriptor deviceDescriptor;
|
||||
|
||||
@@ -58,6 +58,8 @@ extern wgpu::RenderPipeline g_CopyPipeline;
|
||||
extern wgpu::BindGroup g_CopyBindGroup;
|
||||
extern wgpu::Instance g_instance;
|
||||
extern bool g_bcTexturesSupported;
|
||||
// The device was created with TimestampQuery, so passes may carry timestamp writes (gfx::gpu_timing_*).
|
||||
extern bool g_timestampQueriesSupported;
|
||||
|
||||
bool initialize(AuroraBackend backend);
|
||||
void shutdown();
|
||||
|
||||
+20
-1
@@ -143,6 +143,12 @@ suggested for `oculus/touch_controller` and `khr/simple_controller`.
|
||||
provider is registered on Android, Aurora skips the surface present and the
|
||||
desktop mirror copy (`headset_owns_display` in `lib/aurora.cpp`). The game's
|
||||
own render size is unaffected: at `resolution_multiplier = 1` it is 640x528.
|
||||
Since 2026-09-19 an immersive race also stops that native render after the
|
||||
last pass whose EFB copy the eye replays sample (`last_pass_feeding_replay`):
|
||||
the main scene and display copy of a 1280x720 image nobody sees were 4 to
|
||||
6 ms of a 12 ms GPU frame on a Quest 3. A pending CPU readback of an EFB
|
||||
copy or a frame capture still renders the whole image, and menus (the
|
||||
virtual screen) keep it because their eyes are built from that snapshot.
|
||||
- **JNI only on the real thread stack.** Guest threads run on libco stacks
|
||||
inside the SDL thread, and SDL's Android event pump can reach Java (joystick
|
||||
polling, HIDAPI). ART binds JNI transitions to the thread's real stack, so
|
||||
@@ -582,7 +588,7 @@ the app:
|
||||
| `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update |
|
||||
| `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds |
|
||||
| `debug.wiicompiled.inject <n>:<button>` | Presses `a`, `b`, `x`, `y`, `start`, `up`, `down`, `left` or `right` for 12 XR frames each time `<n>` changes. As a Wii Remote, `x`/`y`/`start` are 1/2/+, the directions push the Nunchuk stick, and `home`, `c` and `z` also exist. `panel` presses the settings panel's button (left Y, or both thumbsticks as a gamepad), opening or closing it (see `OPENXR.md`) |
|
||||
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
|
||||
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. A second line gives the GPU time per frame from timestamp queries on every pass (`mono` native render, `eyeL`/`eyeR` replays, `screen`, `panel`, `efbcopy`, `palette`, `peek`, plus `passes-span` from the first pass begin to the last pass end and `between-passes` for copies and idle gaps). The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
|
||||
|
||||
The injector makes headset tests possible with nobody wearing the headset.
|
||||
Keep the display awake, drive the menus, then take a compositor screenshot:
|
||||
@@ -649,6 +655,19 @@ at the start is about 1 ms of game-thread CPU per frame, with the GPU at 85 to
|
||||
89%, so the next steps are on both sides: the guest-code share (translator
|
||||
output quality) and the eye replay's GPU cost.
|
||||
|
||||
The GPU side, measured the same day with per-pass timestamp queries (the second
|
||||
`fpslog` line): on SNES Ghost Valley 2 at `render_scale` 0.5 (840x880 eyes) a
|
||||
stereo frame cost 13.2 ms, of which the native render was 5.7 ms, the eyes 3.5
|
||||
and 3.8, copies and gaps 0.4. That native render is a 1280x720 image nobody
|
||||
sees during an immersive race, so it now stops after the last pass whose EFB
|
||||
copy the eyes sample: `mono` fell to 0.15 ms and a Luigi Circuit start at 0.5
|
||||
renders in 5.5 to 10 ms of GPU per frame. What remains is the headset pacing:
|
||||
with the display at 72 or 90 Hz, each headset frame stays open for the next
|
||||
60 Hz game frame plus the whole encode (`open` 16 ms in the pacing summary),
|
||||
so cycles span one to two display slots and the headset gets 40 to 60 frames
|
||||
per second while the game renders 60. Reworking that pacing (encode the newest
|
||||
sealed frame at once, repeat the layer otherwise) is the next step.
|
||||
|
||||
Verified on device since: the menus on the virtual screen, controller input
|
||||
(the user has driven races), and an immersive Grand Prix start with all 12
|
||||
racers rendering correctly. Not yet verified: stereo comfort and scale,
|
||||
|
||||
Reference in new issue
Block a user