Implement per-pass GPU timing for performance tracking and optimization & fixed mono in Immersive

This commit is contained in:
iChris4 committed 2026-09-19 19:23:57 +02:00
1 parent f0e5e43985
commit 47b59c7294
10 files changed
+344 -5

No files matched your search

+30 -2
View File
@@ -756,6 +756,7 @@ void encode_virtual_screen_eye(wgpu::CommandEncoder& encoder, const webgpu::Pres
.label = eyeIndex == 0 ? "Virtual screen left eye" : "Virtual screen right eye",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::VirtualScreen),
};
{
const auto pass = encoder.BeginRenderPass(&descriptor);
@@ -1274,6 +1275,7 @@ bool present_presentation_job(const PresentationJob& job) {
.label = "Presentation copy pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Present),
};
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(webgpu::g_CopyPipeline);
@@ -1576,6 +1578,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
.label = "Interpolation snapshot pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
};
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
const auto imageWidth = static_cast<float>(image.texture.size.width);
@@ -1627,6 +1630,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
.label = "Snapshot ImGui pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
};
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
pass.SetViewport(0.f, 0.f, static_cast<float>(image.texture.size.width),
@@ -1829,6 +1833,9 @@ void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept {
void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag,
const StereoSceneAnchor& sceneAnchor) {
ZoneScopedN("Seal frame");
// Every pass this cycle encodes, from the seal's probe blits to the final eye, is timed under
// one frame; encode_sealed_frame resolves it on its last submission.
gfx::gpu_timing_begin_frame();
const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{
.label = "Redraw encoder",
};
@@ -1988,7 +1995,18 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
// A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed
// stream would mutate an already-rendered EFB. Render once, then duplicate into the slots.
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo);
//
// On a headset an immersive frame's native render is never presented: the eyes replay the draws
// themselves and only sample the EFB copies it resolves. So it stops after the last pass that
// produces one of those copies (never the display copy), which on a Quest 3 was 4 to 6 ms of a
// 12 ms GPU frame spent on a 1280x720 image nobody saw. A pending CPU readback or a frame
// capture still gets the whole image.
int32_t nativeRenderLastPass = INT32_MAX;
if (headsetOnly && immersiveReplay && !gfx::efb_ram::has_pending() &&
g_captureFrame.load(std::memory_order_acquire) == UINT32_MAX) {
nativeRenderLastPass = gfx::last_pass_feeding_replay(sealedFrame);
}
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo, nativeRenderLastPass);
// The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder;
// completion is harvested in gfx::after_submit, never waited on here.
gfx::efb_ram::encode_async_downloads(encoder);
@@ -2070,7 +2088,9 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
.presentAt = slotPresentDeadline(ctx.interpolatedFrameCount),
.interpolated = false,
});
gfx::gpu_timing_end_frame(encoder);
submitEncodedSlot(encoder, pendingStereoSink ? &*pendingStereoSink : nullptr);
gfx::gpu_timing_after_submit();
// A group that finished encoding past its anchor slides forward by whole display periods, never
// per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period.
@@ -2180,7 +2200,12 @@ void record_frame_telemetry() {
{
// `adb shell setprop debug.wiicompiled.fpslog 1` before launch logs the game's rendered frame rate every five
// seconds. The headset compositor's own log (logcat tag VrApi) repeats frames, so it cannot show this.
static const bool fpsLog = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
static const bool fpsLog = [] {
const bool on = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
// The same switch turns on the per-pass GPU timestamps reported below the frame-rate line.
gfx::gpu_timing_set_enabled(on);
return on;
}();
if (fpsLog) {
static auto windowStart = std::chrono::steady_clock::now();
static uint32_t windowFrames = 0;
@@ -2203,6 +2228,9 @@ void record_frame_telemetry() {
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding",
windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal,
permitWait, prepare, encode);
if (const std::string gpuTiming = gfx::gpu_timing_report(); !gpuTiming.empty()) {
Log.info("{}", gpuTiming);
}
windowStart = now;
windowFrames = 0;
}
+230 -1
View File
@@ -1724,6 +1724,9 @@ struct RenderInvocation {
uint32_t localPlayerCount = 1;
// Inclusive index of the last pass to replay; -1 replays every pass.
int32_t replayLastPass = -1;
// Inclusive index of the last pass that does render work; texture bakes still run for the
// passes after it. See last_pass_feeding_replay.
int32_t renderLastPass = INT32_MAX;
bool finalize = true;
bool replayOnlyEfb = false;
bool skipCopyClears = false;
@@ -1755,6 +1758,11 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
tex_palette_conv::run(cmd, conv);
}
}
if (static_cast<int32_t>(i) > invocation.renderLastPass) {
// Nothing after the last replay-feeding resolve is shown or sampled on a headset; the
// bakes above are all these passes owe the eye replays.
continue;
}
const bool hasRenderWork = passInfo.clearColor || passInfo.clearDepth || !passInfo.commands.empty();
if (i == renderPasses.size() - 1) {
ASSERT(!passInfo.resolveTarget, "Final render pass must not have resolve target");
@@ -1793,11 +1801,16 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
.depthStoreOp = wgpu::StoreOp::Store,
.depthClearValue = passInfo.clearDepthValue,
};
const GpuTimingCategory timingCategory = invocation.stereoEye == 0 ? GpuTimingCategory::EyeLeft
: invocation.stereoEye == 1 ? GpuTimingCategory::EyeRight
: invocation.interpolatedFrame >= 0 ? GpuTimingCategory::Interpolated
: GpuTimingCategory::Mono;
const wgpu::RenderPassDescriptor renderPassDescriptor{
.label = render_pass_label(i),
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.depthStencilAttachment = &depthStencilAttachment,
.timestampWrites = gpu_timing_pass(timingCategory),
};
auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
@@ -1908,15 +1921,28 @@ void seal_frame(SealedFrame& out) noexcept {
g_currentRenderPass = UINT32_MAX;
}
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) {
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize,
int32_t nativeRenderLastPass) {
render_impl(frame.data().passes, cmd,
RenderInvocation{
.interpolatedFrame = interpolatedFrame,
.renderLastPass = nativeRenderLastPass,
.finalize = finalize,
.encodeTextureBakes = interpolatedFrame < 0,
});
}
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept {
const auto& passes = frame.data().passes;
int32_t last = -1;
for (size_t i = 0; i < passes.size(); ++i) {
if (passes[i].resolveTarget && !passes[i].displayCopyResolve) {
last = static_cast<int32_t>(i);
}
}
return last;
}
bool has_late_stereo_replay(const SealedFrame& frame) noexcept {
const auto& data = frame.data().stereo;
return data.generation != 0 && data.generation == g_replayBufferGeneration.load(std::memory_order_acquire) &&
@@ -2028,6 +2054,209 @@ void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize)
}
}
// --- Per-pass GPU timing (see common.hpp) -------------------------------------------------------
namespace {
constexpr uint32_t kGpuTimingSlots = 4;
constexpr uint32_t kGpuTimingPairs = 62;
constexpr uint32_t kGpuTimingQueries = 2 * kGpuTimingPairs;
struct GpuTimingSlot {
wgpu::QuerySet querySet;
wgpu::Buffer resolve;
wgpu::Buffer readback;
std::array<wgpu::PassTimestampWrites, kGpuTimingPairs> writes{};
std::array<GpuTimingCategory, kGpuTimingPairs> categories{};
uint32_t pairs = 0;
bool open = false; // between the frame's begin and end
bool reading = false; // readback in flight or mapped
bool mapped = false; // the callback ran; the encoding thread unmaps on reuse
};
std::atomic<bool> g_gpuTimingEnabled{false};
std::array<GpuTimingSlot, kGpuTimingSlots> g_gpuTimingSlots;
uint32_t g_gpuTimingNextSlot = 0;
int32_t g_gpuTimingCurrent = -1;
bool g_gpuTimingReady = false;
// Guards the totals below and every slot's reading/mapped flags: the map callback may run on
// whichever thread processes Dawn's events.
std::mutex g_gpuTimingMutex;
std::array<uint64_t, static_cast<size_t>(GpuTimingCategory::Count)> g_gpuTimingTotalsNs{};
uint64_t g_gpuTimingSpanNs = 0;
uint32_t g_gpuTimingFrames = 0;
uint32_t g_gpuTimingSkipped = 0;
bool gpu_timing_create_slots() {
if (g_gpuTimingReady) {
return true;
}
if (!webgpu::g_timestampQueriesSupported || !webgpu::g_device) {
return false;
}
for (auto& slot : g_gpuTimingSlots) {
const wgpu::QuerySetDescriptor querySetDescriptor{
.label = "GPU timing queries",
.type = wgpu::QueryType::Timestamp,
.count = kGpuTimingQueries,
};
slot.querySet = webgpu::g_device.CreateQuerySet(&querySetDescriptor);
const wgpu::BufferDescriptor resolveDescriptor{
.label = "GPU timing resolve",
.usage = wgpu::BufferUsage::QueryResolve | wgpu::BufferUsage::CopySrc,
.size = kGpuTimingQueries * sizeof(uint64_t),
};
slot.resolve = webgpu::g_device.CreateBuffer(&resolveDescriptor);
const wgpu::BufferDescriptor readbackDescriptor{
.label = "GPU timing readback",
.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst,
.size = kGpuTimingQueries * sizeof(uint64_t),
};
slot.readback = webgpu::g_device.CreateBuffer(&readbackDescriptor);
}
g_gpuTimingReady = true;
return true;
}
} // namespace
void gpu_timing_set_enabled(bool enabled) noexcept { g_gpuTimingEnabled.store(enabled, std::memory_order_relaxed); }
bool gpu_timing_enabled() noexcept { return g_gpuTimingEnabled.load(std::memory_order_relaxed); }
void gpu_timing_begin_frame() noexcept {
g_gpuTimingCurrent = -1;
if (!gpu_timing_enabled() || !gpu_timing_create_slots()) {
return;
}
const uint32_t index = g_gpuTimingNextSlot;
g_gpuTimingNextSlot = (g_gpuTimingNextSlot + 1) % kGpuTimingSlots;
auto& slot = g_gpuTimingSlots[index];
{
std::lock_guard lock(g_gpuTimingMutex);
if (slot.reading && !slot.mapped) {
++g_gpuTimingSkipped; // the GPU is more than a ring behind; leave this frame untimed
return;
}
if (slot.mapped) {
slot.readback.Unmap();
slot.mapped = false;
}
slot.reading = false;
}
slot.pairs = 0;
slot.open = true;
g_gpuTimingCurrent = static_cast<int32_t>(index);
}
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept {
if (g_gpuTimingCurrent < 0) {
return nullptr;
}
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
if (!slot.open || slot.pairs >= kGpuTimingPairs) {
return nullptr;
}
const uint32_t i = slot.pairs++;
slot.writes[i] = wgpu::PassTimestampWrites{
.querySet = slot.querySet,
.beginningOfPassWriteIndex = 2 * i,
.endOfPassWriteIndex = 2 * i + 1,
};
slot.categories[i] = category;
return &slot.writes[i];
}
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept {
if (g_gpuTimingCurrent < 0) {
return;
}
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
slot.open = false;
if (slot.pairs == 0) {
g_gpuTimingCurrent = -1;
return;
}
const uint32_t queries = 2 * slot.pairs;
encoder.ResolveQuerySet(slot.querySet, 0, queries, slot.resolve, 0);
encoder.CopyBufferToBuffer(slot.resolve, 0, slot.readback, 0, queries * sizeof(uint64_t));
}
void gpu_timing_after_submit() noexcept {
if (g_gpuTimingCurrent < 0) {
return;
}
const uint32_t index = static_cast<uint32_t>(g_gpuTimingCurrent);
g_gpuTimingCurrent = -1;
auto& slot = g_gpuTimingSlots[index];
const uint32_t pairs = slot.pairs;
{
std::lock_guard lock(g_gpuTimingMutex);
slot.reading = true;
slot.mapped = false;
}
slot.readback.MapAsync(
wgpu::MapMode::Read, 0, 2 * pairs * sizeof(uint64_t), wgpu::CallbackMode::AllowSpontaneous,
[index, pairs](wgpu::MapAsyncStatus status, wgpu::StringView) {
auto& slot = g_gpuTimingSlots[index];
std::lock_guard lock(g_gpuTimingMutex);
if (status != wgpu::MapAsyncStatus::Success) {
slot.reading = false;
return;
}
const auto* stamps =
static_cast<const uint64_t*>(slot.readback.GetConstMappedRange(0, 2 * pairs * sizeof(uint64_t)));
if (stamps != nullptr) {
uint64_t first = UINT64_MAX;
uint64_t last = 0;
for (uint32_t i = 0; i < pairs; ++i) {
const uint64_t begin = stamps[2 * i];
const uint64_t end = stamps[2 * i + 1];
if (end < begin) {
continue;
}
g_gpuTimingTotalsNs[static_cast<size_t>(slot.categories[i])] += end - begin;
first = std::min(first, begin);
last = std::max(last, end);
}
if (last > first) {
g_gpuTimingSpanNs += last - first;
}
++g_gpuTimingFrames;
}
slot.mapped = true;
});
}
std::string gpu_timing_report() {
std::lock_guard lock(g_gpuTimingMutex);
if (g_gpuTimingFrames == 0 && g_gpuTimingSkipped == 0) {
return {};
}
static constexpr std::array<const char*, static_cast<size_t>(GpuTimingCategory::Count)> kNames{
"mono", "eyeL", "eyeR", "interp", "screen", "panel", "efbcopy", "palette", "peek", "snapshot", "present"};
std::string text;
if (g_gpuTimingFrames != 0) {
const double frames = g_gpuTimingFrames;
uint64_t sum = 0;
text += fmt::format("GPU ms/frame over {} frames: passes-span={:.2f}", g_gpuTimingFrames,
static_cast<double>(g_gpuTimingSpanNs) / 1e6 / frames);
for (size_t i = 0; i < kNames.size(); ++i) {
if (g_gpuTimingTotalsNs[i] == 0) {
continue;
}
sum += g_gpuTimingTotalsNs[i];
text += fmt::format(" {}={:.2f}", kNames[i], static_cast<double>(g_gpuTimingTotalsNs[i]) / 1e6 / frames);
}
const uint64_t between = g_gpuTimingSpanNs > sum ? g_gpuTimingSpanNs - sum : 0;
text += fmt::format(" between-passes={:.2f}", static_cast<double>(between) / 1e6 / frames);
}
if (g_gpuTimingSkipped != 0) {
text += fmt::format(" (untimed frames: {})", g_gpuTimingSkipped);
}
g_gpuTimingTotalsNs.fill(0);
g_gpuTimingSpanNs = 0;
g_gpuTimingFrames = 0;
g_gpuTimingSkipped = 0;
return text;
}
void after_submit() noexcept {
depth_peek::after_submit();
efb_ram::after_submit();
+42 -1
View File
@@ -8,6 +8,7 @@
#include <cstring>
#include <array>
#include <memory>
#include <string>
#include <type_traits>
#include <utility>
@@ -358,7 +359,14 @@ bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, c
// Encode a sealed frame. Never touches the producer-visible recording state,
// so this may run concurrently with the producer's FIFO drains.
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true);
// `nativeRenderLastPass` limits the passes that do render work (texture bakes still run for every
// pass): a headset never shows an immersive frame's native render, so encode_sealed_frame stops it
// after the last pass whose EFB copy the eye replays sample.
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true,
int32_t nativeRenderLastPass = INT32_MAX);
// Index of the last recorded pass that resolves an EFB copy other than the display copy, or -1
// when no pass does: everything after it exists only for the presented image.
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept;
// Replays only main-EFB passes into one Aurora-owned eye target. Native
// offscreen/EFB-copy passes are consumed from the mono render and are not
@@ -414,6 +422,39 @@ bool is_offscreen() noexcept;
uint32_t get_sample_count() noexcept;
void clear_caches() noexcept;
// Per-pass GPU timing for the frame-rate log. When enabled and the device has TimestampQuery,
// every render or compute pass asks gpu_timing_pass() for timestamp writes under a category; the
// frame's queries are resolved into a small ring of readback buffers and the completed frames'
// durations are summed per category until gpu_timing_report() consumes them. Off by default:
// aurora.cpp enables it together with the Android frame-rate log.
enum class GpuTimingCategory : uint8_t {
Mono, // the native (desktop) render of the recorded GX passes
EyeLeft, // stereo replay of the left eye
EyeRight, // stereo replay of the right eye
Interpolated, // interpolated presentation slots
VirtualScreen, // the 2D virtual screen built for each eye
Panel, // the in-headset settings panel
EfbCopy, // EFB copy format conversions
Palette, // palette (TLUT) texture conversions
DepthPeek, // the depth snapshot compute pass
Snapshot, // presentation snapshot and its ImGui pass
Present, // the desktop presentation copy
Count,
};
void gpu_timing_set_enabled(bool enabled) noexcept;
bool gpu_timing_enabled() noexcept;
// Opens the current frame's query slot; a frame whose slot is still being read back is skipped.
void gpu_timing_begin_frame() noexcept;
// Timestamp writes for one pass of the open frame, or nullptr when timing is off or exhausted.
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept;
// Resolves the open frame's queries on `encoder`, which must be the frame's last submission.
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept;
// After that submission: starts the readback of the resolved queries.
void gpu_timing_after_submit() noexcept;
// Per-frame averages of the frames read back since the last call, formatted for the log, or
// an empty string when nothing was measured.
std::string gpu_timing_report();
namespace tex_palette_conv {
struct ConvRequest;
} // namespace tex_palette_conv
+2
View File
@@ -1,4 +1,5 @@
#include "depth_peek.hpp"
#include "common.hpp"
#include "../dolphin/vi/vi_internal.hpp"
#include "../gx/gx.hpp"
@@ -403,6 +404,7 @@ void encode_frame_snapshot(const wgpu::CommandEncoder& cmd, const wgpu::TextureV
const wgpu::ComputePassDescriptor passDescriptor{
.label = "Depth Peek Compute Pass",
.timestampWrites = gpu_timing_pass(GpuTimingCategory::DepthPeek),
};
const auto pass = cmd.BeginComputePass(&passDescriptor);
pass.SetPipeline(g_pipeline);
+2
View File
@@ -1,4 +1,5 @@
#include "tex_copy_conv.hpp"
#include "common.hpp"
#include "tex_copy_format_contract.hpp"
#include "../internal.hpp"
@@ -586,6 +587,7 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con
.label = "TexCopyConv Pass",
.colorAttachmentCount = colorAttachments.size(),
.colorAttachments = colorAttachments.data(),
.timestampWrites = gpu_timing_pass(GpuTimingCategory::EfbCopy),
};
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(pipeline);
+2
View File
@@ -1,4 +1,5 @@
#include "tex_palette_conv.hpp"
#include "common.hpp"
#include "../internal.hpp"
#include "../webgpu/gpu.hpp"
@@ -249,6 +250,7 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
.label = "TexPaletteConv Pass",
.colorAttachmentCount = colorAttachments.size(),
.colorAttachments = colorAttachments.data(),
.timestampWrites = gpu_timing_pass(GpuTimingCategory::Palette),
};
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(pipeline);
+2
View File
@@ -239,6 +239,7 @@ void composite(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& tar
.label = eyeIndex == 0 ? "Headset panel left eye" : "Headset panel right eye",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
};
const auto pass = encoder.BeginRenderPass(&descriptor);
pass.SetPipeline(state.pipeline);
@@ -294,6 +295,7 @@ wgpu::CommandBuffer prepare(ImDrawData* drawData, float widthFraction) noexcept
.label = "Headset panel ImGui pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
};
bool drawn = false;
{
+12
View File
@@ -80,6 +80,7 @@ wgpu::Instance g_instance;
static wgpu::AdapterInfo g_adapterInfo;
static wgpu::SurfaceCapabilities g_surfaceCapabilities;
bool g_bcTexturesSupported;
bool g_timestampQueriesSupported = false;
// Written by Dawn's device-loss callback and consumed at ordered frame boundaries. Keep the
// callback free of logging, allocation, teardown and renderer state mutation.
static std::atomic_bool g_deviceLost{false};
@@ -702,6 +703,12 @@ bool initialize(AuroraBackend auroraBackend) {
g_bcTexturesSupported = true;
requiredFeatures.push_back(feature);
}
// Per-pass GPU timing for the frame-rate log (gfx::gpu_timing_*). Requesting the feature
// costs nothing until a pass carries timestamp writes.
if (feature == wgpu::FeatureName::TimestampQuery) {
g_timestampQueriesSupported = true;
requiredFeatures.push_back(feature);
}
// The presenter calls device and queue methods while the frame worker encodes, which Dawn only
// supports with this feature; without it the two race inside the device's dynamic uploader.
if (feature == wgpu::FeatureName::ImplicitDeviceSynchronization) {
@@ -769,10 +776,15 @@ bool initialize(AuroraBackend auroraBackend) {
if (g_backendType == wgpu::BackendType::Vulkan) {
enableToggles.push_back("vulkan_monolithic_pipeline_cache");
}
// Dawn quantizes timestamp queries to 100 us for web privacy; the per-pass GPU timing wants
// the raw values.
const std::array<const char*, 1> disableToggles{"timestamp_quantization"};
const wgpu::DawnTogglesDescriptor togglesDescriptor({
.nextInChain = &cacheDescriptor,
.enabledToggleCount = enableToggles.size(),
.enabledToggles = enableToggles.data(),
.disabledToggleCount = g_timestampQueriesSupported ? disableToggles.size() : 0,
.disabledToggles = disableToggles.data(),
});
#endif
wgpu::DeviceDescriptor deviceDescriptor;
+2
View File
@@ -58,6 +58,8 @@ extern wgpu::RenderPipeline g_CopyPipeline;
extern wgpu::BindGroup g_CopyBindGroup;
extern wgpu::Instance g_instance;
extern bool g_bcTexturesSupported;
// The device was created with TimestampQuery, so passes may carry timestamp writes (gfx::gpu_timing_*).
extern bool g_timestampQueriesSupported;
bool initialize(AuroraBackend backend);
void shutdown();