diff --git a/android/QuestGameKit.psm1 b/android/QuestGameKit.psm1 index f61f94a..7377a04 100644 --- a/android/QuestGameKit.psm1 +++ b/android/QuestGameKit.psm1 @@ -539,7 +539,10 @@ function Invoke-QuestGameBuild { } $allObjects = @($sources.Keys | ForEach-Object { $objectsOf[$_] } | ForEach-Object { & $escape $_.Substring($build.Length + 1) }) $rspInputs = ($linkInputs | ForEach-Object { ConvertTo-QuotedArgument $_ }) -join ' ' - [void]$ninjaText.AppendLine("build libmain.so: link $($allObjects -join ' ')`n flags = $(& $escape "$build/link.rsp")`n inputs = $($rspInputs.Replace('$', '$$'))") + # The kit's objects and archives keep their names from one kit export to the next, so they are + # implicit inputs of the link: a game built against an earlier kit is relinked, not reused. + $kitInputs = @($linkInputs | Where-Object { $_.StartsWith($kit) } | ForEach-Object { & $escape $_ }) + [void]$ninjaText.AppendLine("build libmain.so: link $($allObjects -join ' ') | $($kitInputs -join ' ')`n flags = $(& $escape "$build/link.rsp")`n inputs = $($rspInputs.Replace('$', '$$'))") [IO.File]::WriteAllText("$build/build.ninja", $ninjaText.ToString(), (New-Object Text.UTF8Encoding $false)) # Ninja's progress goes to the console, not into this function's return value; its [n/N] diff --git a/android/app/src/main/java/org/wiicompiled/quest/QuestActivity.kt b/android/app/src/main/java/org/wiicompiled/quest/QuestActivity.kt index a9f4901..0a5ead5 100644 --- a/android/app/src/main/java/org/wiicompiled/quest/QuestActivity.kt +++ b/android/app/src/main/java/org/wiicompiled/quest/QuestActivity.kt @@ -97,13 +97,17 @@ class QuestActivity : SDLActivity() { /** * Copies runtime_resources/ from the APK assets into private storage the - * first time this build runs. A stamp file keyed on the version code keeps - * every later launch to one existence check. + * first time this build runs. A stamp file keyed on the version and the + * install time keeps every later launch to one existence check. */ private fun unpackRuntimeResources(): File { val target = File(filesDir, "runtime_resources") val stamp = File(target, ".version") - val expected = "${BuildConfig.VERSION_CODE}:${BuildConfig.VERSION_NAME}" + // The version fields stay the same across sideloaded builds, so the install time is what + // keeps a rebuilt APK from reusing the previous build's bundled pipeline cache. + @Suppress("DEPRECATION") + val installedAt = packageManager.getPackageInfo(packageName, 0).lastUpdateTime + val expected = "${BuildConfig.VERSION_CODE}:${BuildConfig.VERSION_NAME}:$installedAt" if (stamp.isFile && stamp.readText() == expected) { return target } diff --git a/aurora-main/include/aurora/aurora.h b/aurora-main/include/aurora/aurora.h index 2fb7468..1e8fe87 100644 --- a/aurora-main/include/aurora/aurora.h +++ b/aurora-main/include/aurora/aurora.h @@ -254,6 +254,13 @@ bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros); // worker is fully done. Unlike aurora_wait_for_frame_worker(), this is safe in // the gap between aurora_end_frame() and aurora_begin_frame(). void aurora_quiesce_frame_worker(); +// Persists the renderer's pipeline caches now (Dawn's Vulkan pipeline cache, then the queued +// pipeline recipes). It holds the GPU device for the store, so a race would see it as a stall: +// call it at a race exit, when the session loses focus, or before ending the process. +void aurora_store_pipeline_caches(); +// Allows the pipeline compiler to store the caches itself, rate-limited, whenever a first-use +// burst completes. Off by default; enable it while a stall is acceptable, such as in menus. +void aurora_set_pipeline_cache_idle_store(bool allowed); // Absolute schedule for the next sealed frame, on steady_clock: baseNanos anchors the group and // intervalNanos is the period, so slot k of N+1 fires at base + k*interval/(N+1). Zeros clear it. void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos); diff --git a/aurora-main/lib/aurora.cpp b/aurora-main/lib/aurora.cpp index eb2260e..8463028 100644 --- a/aurora-main/lib/aurora.cpp +++ b/aurora-main/lib/aurora.cpp @@ -24,6 +24,9 @@ #include #include "android_debug.hpp" +#ifdef AURORA_ENABLE_GX +#include "gfx/pipeline_cache.hpp" +#endif #include "system_info.hpp" #include "tracy/Tracy.hpp" @@ -2509,6 +2512,18 @@ bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros) { return aurora::wait_for_frame_worker_for(std::chrono::microseconds(timeoutMicros)); } void aurora_quiesce_frame_worker() { aurora::quiesce_frame_worker(); } +void aurora_store_pipeline_caches() { +#ifdef AURORA_ENABLE_GX + aurora::gfx::store_pipeline_caches(); +#endif +} +void aurora_set_pipeline_cache_idle_store(bool allowed) { +#ifdef AURORA_ENABLE_GX + aurora::gfx::set_pipeline_cache_idle_store(allowed); +#else + (void)allowed; +#endif +} void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos) { aurora::g_presentScheduleBaseNanos.store(baseNanos, std::memory_order_release); aurora::g_presentScheduleIntervalNanos.store(intervalNanos, std::memory_order_release); diff --git a/aurora-main/lib/gfx/pipeline_cache.cpp b/aurora-main/lib/gfx/pipeline_cache.cpp index b0c94d4..0038174 100644 --- a/aurora-main/lib/gfx/pipeline_cache.cpp +++ b/aurora-main/lib/gfx/pipeline_cache.cpp @@ -93,6 +93,11 @@ static std::mutex g_pipelineCacheWriterMutex; static std::deque g_pipelineCacheWriteQueue; static absl::flat_hash_set g_pipelineCachePendingWrites; static bool g_pipelineCacheWriterStop = false; +// Set while the writer thread is alive and while it is inside a write transaction, both under +// g_pipelineCacheWriterMutex, so a flush can wait for the queue to reach the database. +static bool g_pipelineCacheWriterRunning = false; +static bool g_pipelineCacheWriterBusy = false; +static std::condition_variable g_pipelineCacheWriterIdleCv; static int g_sdlVfsRegisterResult = SQLITE_ERROR; static SdlVfsSqliteFile* sdl_vfs_file(sqlite3_file* file) { @@ -909,10 +914,13 @@ static void pipeline_cache_writer() { g_pipelineCacheWriterCv.wait(lock, [] { return g_pipelineCacheWriterStop || !g_pipelineCacheWriteQueue.empty(); }); if (g_pipelineCacheWriterStop && g_pipelineCacheWriteQueue.empty()) { + g_pipelineCacheWriterRunning = false; + g_pipelineCacheWriterIdleCv.notify_all(); return; } batch.swap(g_pipelineCacheWriteQueue); g_pipelineCachePendingWrites.clear(); + g_pipelineCacheWriterBusy = true; } bool writeFailed = false; @@ -935,6 +943,15 @@ static void pipeline_cache_writer() { } } + { + std::lock_guard lock{g_pipelineCacheWriterMutex}; + g_pipelineCacheWriterBusy = false; + if (writeFailed) { + g_pipelineCacheWriterRunning = false; + } + g_pipelineCacheWriterIdleCv.notify_all(); + } + if (writeFailed) { pipeline_cache_abort(); return; @@ -948,21 +965,48 @@ static std::atomic_bool g_prewarmActive{false}; static std::chrono::steady_clock::time_point g_prewarmStart{}; static uint32_t g_prewarmCount = 0; +// Storing the monolithic Vulkan pipeline cache holds the device lock for the whole +// vkGetPipelineCacheData, compression and database write, a visible stall mid-race. Outside boot +// prewarm it therefore runs only at moments the host declares safe: on request, through +// store_pipeline_caches at a race exit and before the process ends, and, while the host allows +// idle stores in menus, when a first-use burst drains, at most once per interval. Dawn skips the +// store when no pipeline was created since the last one, but a creation served from the cache +// still counts, so every store rewrites the whole blob; the interval keeps that rare. +static std::mutex g_storeMutex; +static std::atomic_bool g_idleStoreAllowed{false}; +static std::chrono::steady_clock::time_point g_lastStore{}; +constexpr auto kIdleStoreInterval = std::chrono::seconds(120); + static void note_pipeline_queue_drained() { - if (!g_prewarmActive.exchange(false, std::memory_order_acq_rel)) { + if (g_prewarmActive.exchange(false, std::memory_order_acq_rel)) { + { + std::lock_guard lock{g_storeMutex}; + webgpu::serialize_pipeline_caches(); + g_lastStore = std::chrono::steady_clock::now(); + } + const auto elapsed = + std::chrono::duration_cast(std::chrono::steady_clock::now() - g_prewarmStart); + const auto stats = webgpu::blob_cache_stats(); + Log.info("Pipeline prewarm finished: {} pipelines in {:.1f} s (Dawn blob cache: {}/{} hits, {} stores, {:.1f} MiB " + "loaded)", + g_prewarmCount, elapsed.count() / 1000.0, stats.hits, stats.lookups, stats.stores, + static_cast(stats.hitBytes) / (1024.0 * 1024.0)); return; } - // Persist the monolithic Vulkan pipeline cache once after boot prewarm only; doing it on - // every drained burst stalls the device lock mid-race. Later first-use compiles are - // covered by the shutdown serialize. + if (!g_idleStoreAllowed.load(std::memory_order_relaxed)) { + return; + } + // A store in progress on another worker covers this burst too. + std::unique_lock lock{g_storeMutex, std::try_to_lock}; + if (!lock.owns_lock()) { + return; + } + const auto now = std::chrono::steady_clock::now(); + if (now - g_lastStore < kIdleStoreInterval) { + return; + } + g_lastStore = now; webgpu::serialize_pipeline_caches(); - const auto elapsed = - std::chrono::duration_cast(std::chrono::steady_clock::now() - g_prewarmStart); - const auto stats = webgpu::blob_cache_stats(); - Log.info("Pipeline prewarm finished: {} pipelines in {:.1f} s (Dawn blob cache: {}/{} hits, {} stores, {:.1f} MiB " - "loaded)", - g_prewarmCount, elapsed.count() / 1000.0, stats.hits, stats.lookups, stats.stores, - static_cast(stats.hitBytes) / (1024.0 * 1024.0)); } static void compile_pending_pipeline(PendingPipeline pending) { @@ -1130,10 +1174,28 @@ static void start_pipeline_cache_writer() { return; } - g_pipelineCacheWriterStop = false; + { + std::lock_guard lock{g_pipelineCacheWriterMutex}; + g_pipelineCacheWriterStop = false; + g_pipelineCacheWriterRunning = true; + g_pipelineCacheWriterBusy = false; + } g_pipelineCacheWriterThread = std::thread(pipeline_cache_writer); } +// Waits, bounded, until every queued recipe row has reached the database: the caller is about to +// end the process, or wants the store it just made to be complete on disk. +static void flush_pipeline_cache_writes() { + std::unique_lock lock{g_pipelineCacheWriterMutex}; + if (!g_pipelineCacheWriterRunning) { + return; + } + g_pipelineCacheWriterCv.notify_one(); + g_pipelineCacheWriterIdleCv.wait_for(lock, std::chrono::seconds(2), [] { + return !g_pipelineCacheWriterRunning || (g_pipelineCacheWriteQueue.empty() && !g_pipelineCacheWriterBusy); + }); +} + static void stop_pipeline_cache_writer() { if (g_pipelineCacheWriterThread.joinable()) { { @@ -1146,10 +1208,32 @@ static void stop_pipeline_cache_writer() { g_pipelineCacheWriterStop = false; } + g_pipelineCacheWriterRunning = false; + g_pipelineCacheWriterBusy = false; g_pipelineCacheWriteQueue.clear(); g_pipelineCachePendingWrites.clear(); } +void store_pipeline_caches() { + const uint64_t storesBefore = webgpu::blob_cache_stats().stores; + const auto start = std::chrono::steady_clock::now(); + { + std::lock_guard lock{g_storeMutex}; + webgpu::serialize_pipeline_caches(); + g_lastStore = std::chrono::steady_clock::now(); + } + flush_pipeline_cache_writes(); + const uint64_t stored = webgpu::blob_cache_stats().stores - storesBefore; + if (stored != 0) { + const auto elapsed = std::chrono::duration_cast(std::chrono::steady_clock::now() - start); + Log.info("Stored the pipeline caches in {} ms: {} new Dawn blob cache entries", elapsed.count(), stored); + } +} + +void set_pipeline_cache_idle_store(bool allowed) noexcept { + g_idleStoreAllowed.store(allowed, std::memory_order_relaxed); +} + template <> PipelineRef find_pipeline(ShaderType type, const clear::PipelineConfig& config, NewPipelineCallback&& cb) { return find_pipeline_impl(type, config, std::move(cb), true, std::nullopt); @@ -1163,6 +1247,8 @@ PipelineRef find_pipeline(ShaderType type, const gx::PipelineConfig& config, New void initialize_pipeline_cache() { g_pipelineCacheBroken = false; g_pipelineCacheWriterStop = false; + g_idleStoreAllowed.store(false, std::memory_order_relaxed); + g_lastStore = {}; g_pipelineFrameActive = false; g_pipelineThreadEnd = false; g_activeBackgroundPipelineWorkers = 0; diff --git a/aurora-main/lib/gfx/pipeline_cache.hpp b/aurora-main/lib/gfx/pipeline_cache.hpp index 737e531..c374f49 100644 --- a/aurora-main/lib/gfx/pipeline_cache.hpp +++ b/aurora-main/lib/gfx/pipeline_cache.hpp @@ -23,6 +23,13 @@ void end_pipeline_frame(); void set_skip_unready_pipelines(bool enabled) noexcept; bool skip_unready_pipelines() noexcept; uint32_t queued_pipeline_count() noexcept; +// Persists the pipeline caches now: Dawn's Vulkan pipeline cache when a pipeline was compiled +// since the last store, then every queued recipe row. Holds the device for the store, so call it +// where a stall is invisible (a race exit, process exit). +void store_pipeline_caches(); +// Lets a drained first-use burst store the caches itself, rate-limited. Off by default; a host +// turns it on while a stall is acceptable (menus) and off again for a race. +void set_pipeline_cache_idle_store(bool allowed) noexcept; template PipelineRef find_pipeline(ShaderType type, const Config& config, NewPipelineCallback&& cb); diff --git a/aurora-main/lib/webgpu/gpu.cpp b/aurora-main/lib/webgpu/gpu.cpp index 1e98e59..58ed9ec 100644 --- a/aurora-main/lib/webgpu/gpu.cpp +++ b/aurora-main/lib/webgpu/gpu.cpp @@ -913,10 +913,16 @@ void fail_if_device_lost() noexcept { } void serialize_pipeline_caches() noexcept { -#if defined(WEBGPU_DAWN) && defined(_WIN32) +#if defined(WEBGPU_DAWN) + // Only the Vulkan backend keeps a monolithic VkPipelineCache (toggle above). Dawn writes it to + // the blob cache from PerformIdleTasks, and only when a pipeline was compiled since the last + // store, so calling this when nothing changed is cheap. if (!g_device || g_backendType != wgpu::BackendType::Vulkan) { return; } +#if defined(_WIN32) + // The Windows product links the Dawn DLL from llvm-mingw, which cannot call the exported C++ + // symbol directly; resolve its MSVC-mangled name instead. using PerformIdleTasksFn = void(*)(const wgpu::Device*); static const auto performIdleTasks = []() -> PerformIdleTasksFn { const HMODULE dawnModule = GetModuleHandleW(L"webgpu_dawn.dll"); @@ -929,6 +935,10 @@ void serialize_pipeline_caches() noexcept { if (performIdleTasks != nullptr) { performIdleTasks(&g_device); } +#elif !defined(__MINGW32__) + // Static Dawn (Android, Linux): DawnNative.h is included above. + dawn::native::PerformIdleTasks(g_device); +#endif #endif } diff --git a/docs/quest-port.md b/docs/quest-port.md index 78023d0..867dec4 100644 --- a/docs/quest-port.md +++ b/docs/quest-port.md @@ -518,6 +518,7 @@ Bring-up fixes that only a device could reveal: | Link error on `Android_LockActivityMutex` | SDL's activity mutex is not exported | Aurora-owned mutex plus the `QuestSurface` bracket (above) | | Crypto++ `cpu-features.h` not found | The NDK ships cpu-features as source | Compiled into `mkw_cryptopp` on Android | | Exploded racers and menu characters; smeared movie panels in the menus; then, once those were fixed, damaged eyes and slightly misplaced detail on characters | The Adreno 740 driver reads the wrong bytes when the shader multiplies an index by a stride that is not a multiple of 4. That covers the vertex fetch (`ubuf.vtx_start + vidx * stride + offset`) and indexed array reads (`array_start + index * stride`, e.g. 6-byte S16 normals). GX packs both byte-tight, so skinned models (a 1-byte `PNMTXIDX` first, stride 7) broke everywhere | Android pads every uploaded vertex and every indexed-array element to a 4-byte stride (`padded_upload_stride` in `lib/gx/gx.cpp`). Offsets inside a vertex or element are unchanged, and desktop is unchanged. **Fixed, headset-verified 2026-09-16** at character select and a Grand Prix start | +| Every launch recompiled every shader: a 14 to 34 s prewarm, and the Dawn blob cache reporting exactly one miss and no stores | Dawn's monolithic Vulkan pipeline cache is only written by `PerformIdleTasks`, which `gpu.cpp` resolved through the Windows DLL alone, and the quit path ends the process without `aurora_shutdown`, so nothing compiled after prewarm was kept either | Static Dawn calls it directly; `aurora_store_pipeline_caches` runs at a race exit, when the session loses focus and on the quit path, and the compiler stores idle bursts itself while the headset shows the virtual screen. The unpacked `initial_pipeline_cache.db` is also refreshed per APK install now | How the explosion was isolated, so the next Adreno rendering bug starts further ahead: @@ -607,8 +608,9 @@ or `EndAccess` errors); a black mirror too points at Aurora itself. extension negotiation, Dawn's begin/end layout reporting for AHardwareBuffer imports, and swapchain format choice (`R8G8B8A8_SRGB` is expected). - **Performance.** The desktop product targets x86-64-v3; nothing has been - profiled on the XR2. Expect shader compilation stalls on first run (Aurora's - pipeline cache is bundled) and start with `render_scale` below 1.0 if the + profiled on the XR2. The first run compiles every bundled pipeline recipe + (about half a minute); later runs load Dawn's pipeline cache from `Cache/` + next to `DATA`. Start with `render_scale` below 1.0 if the compositor reports missed frames. `XR_FB_foveation` is not used yet. - **Lifecycle.** Backgrounding (the Quest menu, guardian) pauses the session through the ordinary `STOPPING`/`READY` events; SDL's Android surface loss is diff --git a/runtime/include/aurora_events.h b/runtime/include/aurora_events.h index 6be4725..403a9ad 100644 --- a/runtime/include/aurora_events.h +++ b/runtime/include/aurora_events.h @@ -61,6 +61,10 @@ inline void Flush(bool force = false) { [[noreturn]] inline void ExitForAuroraWindowClose() noexcept { settings_overlay::ReleaseControllers(); WindowPlacementPersistence::Flush(true); + // Ending the process here skips aurora_shutdown, which is where Dawn's Vulkan pipeline cache + // and the queued pipeline recipes would otherwise reach disk. Without this store every + // session recompiled what it had compiled after boot prewarm (a long stall on the Quest). + aurora_store_pipeline_caches(); #if defined(_WIN32) ::ExitProcess(0); #else diff --git a/runtime/src/vr/openxr_integration.cpp b/runtime/src/vr/openxr_integration.cpp index 49ade39..d3ed506 100644 --- a/runtime/src/vr/openxr_integration.cpp +++ b/runtime/src/vr/openxr_integration.cpp @@ -637,6 +637,8 @@ private: } #endif bool fatal = false; + bool store_gate_set = false; + bool store_gate_immersive = false; bool presentation_logged = false; VRPresentationMode logged_presentation = VRPresentationMode::Desktop; uint32_t presentation_log_count = 0; @@ -655,6 +657,9 @@ private: session_was_active_ = session_active; if (!session_active) { ResetTrackingOrigin(); + // Nothing is displayed while the session is not running (the system menu, + // the headset taken off), so the stall of a cache store is invisible here. + aurora_store_pipeline_caches(); } } if (events == OpenXREventStatus::ExitRequested) { @@ -706,6 +711,19 @@ private: presentation.quad_distance_meters = policy.config.hud_distance_meters; presentation.quad_width_meters = policy.config.hud_width_meters; + // Pipeline caches are stored where their stall is invisible: once when a race ends, + // and by the compiler itself while the headset shows the virtual screen. Never + // mid-race. + if (!store_gate_set || immersive != store_gate_immersive) { + const bool left_race = store_gate_set && store_gate_immersive && !immersive; + store_gate_set = true; + store_gate_immersive = immersive; + aurora_set_pipeline_cache_idle_store(!immersive); + if (left_race) { + aurora_store_pipeline_caches(); + } + } + // Updating this on the owner thread also confines retained replay to // validated race content. The provider checks policy tags again. const uint32_t interpolation_target = frame_interpolation_fps_.load(std::memory_order_relaxed); @@ -841,6 +859,7 @@ private: } SetInterpolationActive(false); + aurora_set_pipeline_cache_idle_store(false); running_.store(false, std::memory_order_release); MkwVRPolicySetSessionActive(false); if (!stop_.load(std::memory_order_acquire)) {