Quest Fixed Vulkan pipeline cache management and storage mechanisms

This commit is contained in:
iChris4 committed 2026-09-18 23:38:55 +02:00
1 parent bf3b53555b
commit d4a3f0b350
10 files changed
+176 -19

No files matched your search

+4 -1
View File
@@ -539,7 +539,10 @@ function Invoke-QuestGameBuild {
}
$allObjects = @($sources.Keys | ForEach-Object { $objectsOf[$_] } | ForEach-Object { & $escape $_.Substring($build.Length + 1) })
$rspInputs = ($linkInputs | ForEach-Object { ConvertTo-QuotedArgument $_ }) -join ' '
[void]$ninjaText.AppendLine("build libmain.so: link $($allObjects -join ' ')`n flags = $(& $escape "$build/link.rsp")`n inputs = $($rspInputs.Replace('$', '$$'))")
# The kit's objects and archives keep their names from one kit export to the next, so they are
# implicit inputs of the link: a game built against an earlier kit is relinked, not reused.
$kitInputs = @($linkInputs | Where-Object { $_.StartsWith($kit) } | ForEach-Object { & $escape $_ })
[void]$ninjaText.AppendLine("build libmain.so: link $($allObjects -join ' ') | $($kitInputs -join ' ')`n flags = $(& $escape "$build/link.rsp")`n inputs = $($rspInputs.Replace('$', '$$'))")
[IO.File]::WriteAllText("$build/build.ninja", $ninjaText.ToString(), (New-Object Text.UTF8Encoding $false))
# Ninja's progress goes to the console, not into this function's return value; its [n/N]
@@ -97,13 +97,17 @@ class QuestActivity : SDLActivity() {
/**
* Copies runtime_resources/ from the APK assets into private storage the
* first time this build runs. A stamp file keyed on the version code keeps
* every later launch to one existence check.
* first time this build runs. A stamp file keyed on the version and the
* install time keeps every later launch to one existence check.
*/
private fun unpackRuntimeResources(): File {
val target = File(filesDir, "runtime_resources")
val stamp = File(target, ".version")
val expected = "${BuildConfig.VERSION_CODE}:${BuildConfig.VERSION_NAME}"
// The version fields stay the same across sideloaded builds, so the install time is what
// keeps a rebuilt APK from reusing the previous build's bundled pipeline cache.
@Suppress("DEPRECATION")
val installedAt = packageManager.getPackageInfo(packageName, 0).lastUpdateTime
val expected = "${BuildConfig.VERSION_CODE}:${BuildConfig.VERSION_NAME}:$installedAt"
if (stamp.isFile && stamp.readText() == expected) {
return target
}
+7
View File
@@ -254,6 +254,13 @@ bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros);
// worker is fully done. Unlike aurora_wait_for_frame_worker(), this is safe in
// the gap between aurora_end_frame() and aurora_begin_frame().
void aurora_quiesce_frame_worker();
// Persists the renderer's pipeline caches now (Dawn's Vulkan pipeline cache, then the queued
// pipeline recipes). It holds the GPU device for the store, so a race would see it as a stall:
// call it at a race exit, when the session loses focus, or before ending the process.
void aurora_store_pipeline_caches();
// Allows the pipeline compiler to store the caches itself, rate-limited, whenever a first-use
// burst completes. Off by default; enable it while a stall is acceptable, such as in menus.
void aurora_set_pipeline_cache_idle_store(bool allowed);
// Absolute schedule for the next sealed frame, on steady_clock: baseNanos anchors the group and
// intervalNanos is the period, so slot k of N+1 fires at base + k*interval/(N+1). Zeros clear it.
void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos);
+15
View File
@@ -24,6 +24,9 @@
#include <magic_enum.hpp>
#include "android_debug.hpp"
#ifdef AURORA_ENABLE_GX
#include "gfx/pipeline_cache.hpp"
#endif
#include "system_info.hpp"
#include "tracy/Tracy.hpp"
@@ -2509,6 +2512,18 @@ bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros) {
return aurora::wait_for_frame_worker_for(std::chrono::microseconds(timeoutMicros));
}
void aurora_quiesce_frame_worker() { aurora::quiesce_frame_worker(); }
void aurora_store_pipeline_caches() {
#ifdef AURORA_ENABLE_GX
aurora::gfx::store_pipeline_caches();
#endif
}
void aurora_set_pipeline_cache_idle_store(bool allowed) {
#ifdef AURORA_ENABLE_GX
aurora::gfx::set_pipeline_cache_idle_store(allowed);
#else
(void)allowed;
#endif
}
void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos) {
aurora::g_presentScheduleBaseNanos.store(baseNanos, std::memory_order_release);
aurora::g_presentScheduleIntervalNanos.store(intervalNanos, std::memory_order_release);
+98 -12
View File
@@ -93,6 +93,11 @@ static std::mutex g_pipelineCacheWriterMutex;
static std::deque<PipelineCacheWrite> g_pipelineCacheWriteQueue;
static absl::flat_hash_set<PipelineRef> g_pipelineCachePendingWrites;
static bool g_pipelineCacheWriterStop = false;
// Set while the writer thread is alive and while it is inside a write transaction, both under
// g_pipelineCacheWriterMutex, so a flush can wait for the queue to reach the database.
static bool g_pipelineCacheWriterRunning = false;
static bool g_pipelineCacheWriterBusy = false;
static std::condition_variable g_pipelineCacheWriterIdleCv;
static int g_sdlVfsRegisterResult = SQLITE_ERROR;
static SdlVfsSqliteFile* sdl_vfs_file(sqlite3_file* file) {
@@ -909,10 +914,13 @@ static void pipeline_cache_writer() {
g_pipelineCacheWriterCv.wait(lock,
[] { return g_pipelineCacheWriterStop || !g_pipelineCacheWriteQueue.empty(); });
if (g_pipelineCacheWriterStop && g_pipelineCacheWriteQueue.empty()) {
g_pipelineCacheWriterRunning = false;
g_pipelineCacheWriterIdleCv.notify_all();
return;
}
batch.swap(g_pipelineCacheWriteQueue);
g_pipelineCachePendingWrites.clear();
g_pipelineCacheWriterBusy = true;
}
bool writeFailed = false;
@@ -935,6 +943,15 @@ static void pipeline_cache_writer() {
}
}
{
std::lock_guard lock{g_pipelineCacheWriterMutex};
g_pipelineCacheWriterBusy = false;
if (writeFailed) {
g_pipelineCacheWriterRunning = false;
}
g_pipelineCacheWriterIdleCv.notify_all();
}
if (writeFailed) {
pipeline_cache_abort();
return;
@@ -948,21 +965,48 @@ static std::atomic_bool g_prewarmActive{false};
static std::chrono::steady_clock::time_point g_prewarmStart{};
static uint32_t g_prewarmCount = 0;
// Storing the monolithic Vulkan pipeline cache holds the device lock for the whole
// vkGetPipelineCacheData, compression and database write, a visible stall mid-race. Outside boot
// prewarm it therefore runs only at moments the host declares safe: on request, through
// store_pipeline_caches at a race exit and before the process ends, and, while the host allows
// idle stores in menus, when a first-use burst drains, at most once per interval. Dawn skips the
// store when no pipeline was created since the last one, but a creation served from the cache
// still counts, so every store rewrites the whole blob; the interval keeps that rare.
static std::mutex g_storeMutex;
static std::atomic_bool g_idleStoreAllowed{false};
static std::chrono::steady_clock::time_point g_lastStore{};
constexpr auto kIdleStoreInterval = std::chrono::seconds(120);
static void note_pipeline_queue_drained() {
if (!g_prewarmActive.exchange(false, std::memory_order_acq_rel)) {
if (g_prewarmActive.exchange(false, std::memory_order_acq_rel)) {
{
std::lock_guard lock{g_storeMutex};
webgpu::serialize_pipeline_caches();
g_lastStore = std::chrono::steady_clock::now();
}
const auto elapsed =
std::chrono::duration_cast<std::chrono::milliseconds>(std::chrono::steady_clock::now() - g_prewarmStart);
const auto stats = webgpu::blob_cache_stats();
Log.info("Pipeline prewarm finished: {} pipelines in {:.1f} s (Dawn blob cache: {}/{} hits, {} stores, {:.1f} MiB "
"loaded)",
g_prewarmCount, elapsed.count() / 1000.0, stats.hits, stats.lookups, stats.stores,
static_cast<double>(stats.hitBytes) / (1024.0 * 1024.0));
return;
}
// Persist the monolithic Vulkan pipeline cache once after boot prewarm only; doing it on
// every drained burst stalls the device lock mid-race. Later first-use compiles are
// covered by the shutdown serialize.
if (!g_idleStoreAllowed.load(std::memory_order_relaxed)) {
return;
}
// A store in progress on another worker covers this burst too.
std::unique_lock lock{g_storeMutex, std::try_to_lock};
if (!lock.owns_lock()) {
return;
}
const auto now = std::chrono::steady_clock::now();
if (now - g_lastStore < kIdleStoreInterval) {
return;
}
g_lastStore = now;
webgpu::serialize_pipeline_caches();
const auto elapsed =
std::chrono::duration_cast<std::chrono::milliseconds>(std::chrono::steady_clock::now() - g_prewarmStart);
const auto stats = webgpu::blob_cache_stats();
Log.info("Pipeline prewarm finished: {} pipelines in {:.1f} s (Dawn blob cache: {}/{} hits, {} stores, {:.1f} MiB "
"loaded)",
g_prewarmCount, elapsed.count() / 1000.0, stats.hits, stats.lookups, stats.stores,
static_cast<double>(stats.hitBytes) / (1024.0 * 1024.0));
}
static void compile_pending_pipeline(PendingPipeline pending) {
@@ -1130,10 +1174,28 @@ static void start_pipeline_cache_writer() {
return;
}
g_pipelineCacheWriterStop = false;
{
std::lock_guard lock{g_pipelineCacheWriterMutex};
g_pipelineCacheWriterStop = false;
g_pipelineCacheWriterRunning = true;
g_pipelineCacheWriterBusy = false;
}
g_pipelineCacheWriterThread = std::thread(pipeline_cache_writer);
}
// Waits, bounded, until every queued recipe row has reached the database: the caller is about to
// end the process, or wants the store it just made to be complete on disk.
static void flush_pipeline_cache_writes() {
std::unique_lock lock{g_pipelineCacheWriterMutex};
if (!g_pipelineCacheWriterRunning) {
return;
}
g_pipelineCacheWriterCv.notify_one();
g_pipelineCacheWriterIdleCv.wait_for(lock, std::chrono::seconds(2), [] {
return !g_pipelineCacheWriterRunning || (g_pipelineCacheWriteQueue.empty() && !g_pipelineCacheWriterBusy);
});
}
static void stop_pipeline_cache_writer() {
if (g_pipelineCacheWriterThread.joinable()) {
{
@@ -1146,10 +1208,32 @@ static void stop_pipeline_cache_writer() {
g_pipelineCacheWriterStop = false;
}
g_pipelineCacheWriterRunning = false;
g_pipelineCacheWriterBusy = false;
g_pipelineCacheWriteQueue.clear();
g_pipelineCachePendingWrites.clear();
}
void store_pipeline_caches() {
const uint64_t storesBefore = webgpu::blob_cache_stats().stores;
const auto start = std::chrono::steady_clock::now();
{
std::lock_guard lock{g_storeMutex};
webgpu::serialize_pipeline_caches();
g_lastStore = std::chrono::steady_clock::now();
}
flush_pipeline_cache_writes();
const uint64_t stored = webgpu::blob_cache_stats().stores - storesBefore;
if (stored != 0) {
const auto elapsed = std::chrono::duration_cast<std::chrono::milliseconds>(std::chrono::steady_clock::now() - start);
Log.info("Stored the pipeline caches in {} ms: {} new Dawn blob cache entries", elapsed.count(), stored);
}
}
void set_pipeline_cache_idle_store(bool allowed) noexcept {
g_idleStoreAllowed.store(allowed, std::memory_order_relaxed);
}
template <>
PipelineRef find_pipeline(ShaderType type, const clear::PipelineConfig& config, NewPipelineCallback&& cb) {
return find_pipeline_impl(type, config, std::move(cb), true, std::nullopt);
@@ -1163,6 +1247,8 @@ PipelineRef find_pipeline(ShaderType type, const gx::PipelineConfig& config, New
void initialize_pipeline_cache() {
g_pipelineCacheBroken = false;
g_pipelineCacheWriterStop = false;
g_idleStoreAllowed.store(false, std::memory_order_relaxed);
g_lastStore = {};
g_pipelineFrameActive = false;
g_pipelineThreadEnd = false;
g_activeBackgroundPipelineWorkers = 0;
+7
View File
@@ -23,6 +23,13 @@ void end_pipeline_frame();
void set_skip_unready_pipelines(bool enabled) noexcept;
bool skip_unready_pipelines() noexcept;
uint32_t queued_pipeline_count() noexcept;
// Persists the pipeline caches now: Dawn's Vulkan pipeline cache when a pipeline was compiled
// since the last store, then every queued recipe row. Holds the device for the store, so call it
// where a stall is invisible (a race exit, process exit).
void store_pipeline_caches();
// Lets a drained first-use burst store the caches itself, rate-limited. Off by default; a host
// turns it on while a stall is acceptable (menus) and off again for a race.
void set_pipeline_cache_idle_store(bool allowed) noexcept;
template <typename Config>
PipelineRef find_pipeline(ShaderType type, const Config& config, NewPipelineCallback&& cb);
+11 -1
View File
@@ -913,10 +913,16 @@ void fail_if_device_lost() noexcept {
}
void serialize_pipeline_caches() noexcept {
#if defined(WEBGPU_DAWN) && defined(_WIN32)
#if defined(WEBGPU_DAWN)
// Only the Vulkan backend keeps a monolithic VkPipelineCache (toggle above). Dawn writes it to
// the blob cache from PerformIdleTasks, and only when a pipeline was compiled since the last
// store, so calling this when nothing changed is cheap.
if (!g_device || g_backendType != wgpu::BackendType::Vulkan) {
return;
}
#if defined(_WIN32)
// The Windows product links the Dawn DLL from llvm-mingw, which cannot call the exported C++
// symbol directly; resolve its MSVC-mangled name instead.
using PerformIdleTasksFn = void(*)(const wgpu::Device*);
static const auto performIdleTasks = []() -> PerformIdleTasksFn {
const HMODULE dawnModule = GetModuleHandleW(L"webgpu_dawn.dll");
@@ -929,6 +935,10 @@ void serialize_pipeline_caches() noexcept {
if (performIdleTasks != nullptr) {
performIdleTasks(&g_device);
}
#elif !defined(__MINGW32__)
// Static Dawn (Android, Linux): DawnNative.h is included above.
dawn::native::PerformIdleTasks(g_device);
#endif
#endif
}
+4 -2
View File
@@ -518,6 +518,7 @@ Bring-up fixes that only a device could reveal:
| Link error on `Android_LockActivityMutex` | SDL's activity mutex is not exported | Aurora-owned mutex plus the `QuestSurface` bracket (above) |
| Crypto++ `cpu-features.h` not found | The NDK ships cpu-features as source | Compiled into `mkw_cryptopp` on Android |
| Exploded racers and menu characters; smeared movie panels in the menus; then, once those were fixed, damaged eyes and slightly misplaced detail on characters | The Adreno 740 driver reads the wrong bytes when the shader multiplies an index by a stride that is not a multiple of 4. That covers the vertex fetch (`ubuf.vtx_start + vidx * stride + offset`) and indexed array reads (`array_start + index * stride`, e.g. 6-byte S16 normals). GX packs both byte-tight, so skinned models (a 1-byte `PNMTXIDX` first, stride 7) broke everywhere | Android pads every uploaded vertex and every indexed-array element to a 4-byte stride (`padded_upload_stride` in `lib/gx/gx.cpp`). Offsets inside a vertex or element are unchanged, and desktop is unchanged. **Fixed, headset-verified 2026-09-16** at character select and a Grand Prix start |
| Every launch recompiled every shader: a 14 to 34 s prewarm, and the Dawn blob cache reporting exactly one miss and no stores | Dawn's monolithic Vulkan pipeline cache is only written by `PerformIdleTasks`, which `gpu.cpp` resolved through the Windows DLL alone, and the quit path ends the process without `aurora_shutdown`, so nothing compiled after prewarm was kept either | Static Dawn calls it directly; `aurora_store_pipeline_caches` runs at a race exit, when the session loses focus and on the quit path, and the compiler stores idle bursts itself while the headset shows the virtual screen. The unpacked `initial_pipeline_cache.db` is also refreshed per APK install now |
How the explosion was isolated, so the next Adreno rendering bug starts further
ahead:
@@ -607,8 +608,9 @@ or `EndAccess` errors); a black mirror too points at Aurora itself.
extension negotiation, Dawn's begin/end layout reporting for AHardwareBuffer
imports, and swapchain format choice (`R8G8B8A8_SRGB` is expected).
- **Performance.** The desktop product targets x86-64-v3; nothing has been
profiled on the XR2. Expect shader compilation stalls on first run (Aurora's
pipeline cache is bundled) and start with `render_scale` below 1.0 if the
profiled on the XR2. The first run compiles every bundled pipeline recipe
(about half a minute); later runs load Dawn's pipeline cache from `Cache/`
next to `DATA`. Start with `render_scale` below 1.0 if the
compositor reports missed frames. `XR_FB_foveation` is not used yet.
- **Lifecycle.** Backgrounding (the Quest menu, guardian) pauses the session
through the ordinary `STOPPING`/`READY` events; SDL's Android surface loss is
+4
View File
@@ -61,6 +61,10 @@ inline void Flush(bool force = false) {
[[noreturn]] inline void ExitForAuroraWindowClose() noexcept {
settings_overlay::ReleaseControllers();
WindowPlacementPersistence::Flush(true);
// Ending the process here skips aurora_shutdown, which is where Dawn's Vulkan pipeline cache
// and the queued pipeline recipes would otherwise reach disk. Without this store every
// session recompiled what it had compiled after boot prewarm (a long stall on the Quest).
aurora_store_pipeline_caches();
#if defined(_WIN32)
::ExitProcess(0);
#else
+19
View File
@@ -637,6 +637,8 @@ private:
}
#endif
bool fatal = false;
bool store_gate_set = false;
bool store_gate_immersive = false;
bool presentation_logged = false;
VRPresentationMode logged_presentation = VRPresentationMode::Desktop;
uint32_t presentation_log_count = 0;
@@ -655,6 +657,9 @@ private:
session_was_active_ = session_active;
if (!session_active) {
ResetTrackingOrigin();
// Nothing is displayed while the session is not running (the system menu,
// the headset taken off), so the stall of a cache store is invisible here.
aurora_store_pipeline_caches();
}
}
if (events == OpenXREventStatus::ExitRequested) {
@@ -706,6 +711,19 @@ private:
presentation.quad_distance_meters = policy.config.hud_distance_meters;
presentation.quad_width_meters = policy.config.hud_width_meters;
// Pipeline caches are stored where their stall is invisible: once when a race ends,
// and by the compiler itself while the headset shows the virtual screen. Never
// mid-race.
if (!store_gate_set || immersive != store_gate_immersive) {
const bool left_race = store_gate_set && store_gate_immersive && !immersive;
store_gate_set = true;
store_gate_immersive = immersive;
aurora_set_pipeline_cache_idle_store(!immersive);
if (left_race) {
aurora_store_pipeline_caches();
}
}
// Updating this on the owner thread also confines retained replay to
// validated race content. The provider checks policy tags again.
const uint32_t interpolation_target = frame_interpolation_fps_.load(std::memory_order_relaxed);
@@ -841,6 +859,7 @@ private:
}
SetInterpolationActive(false);
aurora_set_pipeline_cache_idle_store(false);
running_.store(false, std::memory_order_release);
MkwVRPolicySetSessionActive(false);
if (!stop_.load(std::memory_order_acquire)) {