From 1a4b0589f4bf2a13947a018d6f178bae255423c4 Mon Sep 17 00:00:00 2001 From: iChris4 Date: Wed, 23 Sep 2026 23:53:51 +0200 Subject: [PATCH] Implement background pipeline cache storage --- aurora-main/include/aurora/aurora.h | 8 +- aurora-main/lib/aurora.cpp | 5 + aurora-main/lib/gfx/pipeline_cache.cpp | 78 ++++++- aurora-main/lib/gfx/pipeline_cache.hpp | 10 +- aurora-main/lib/webgpu/gpu.hpp | 5 + aurora-main/lib/webgpu/gpu_cache.cpp | 270 ++++++++++++++++++++----- aurora-main/tests/CMakeLists.txt | 14 ++ aurora-main/tests/gpu_cache_test.cpp | 159 +++++++++++++++ docs/quest-port.md | 1 + runtime/src/vr/openxr_integration.cpp | 11 +- 10 files changed, 492 insertions(+), 69 deletions(-) create mode 100644 aurora-main/tests/gpu_cache_test.cpp diff --git a/aurora-main/include/aurora/aurora.h b/aurora-main/include/aurora/aurora.h index a4f19d5..089a9e6 100644 --- a/aurora-main/include/aurora/aurora.h +++ b/aurora-main/include/aurora/aurora.h @@ -334,9 +334,13 @@ bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros); // the gap between aurora_end_frame() and aurora_begin_frame(). void aurora_quiesce_frame_worker(); // Persists the renderer's pipeline caches now (Dawn's Vulkan pipeline cache, then the queued -// pipeline recipes). It holds the GPU device for the store, so a race would see it as a stall: -// call it at a race exit, when the session loses focus, or before ending the process. +// pipeline recipes) and returns once they are on disk. Dawn holds the GPU device while it +// serializes its cache, so a race would see that part as a stall: call it at a race exit, when +// the session loses focus, or before ending the process. void aurora_store_pipeline_caches(); +// The same store on a background thread, returning at once, for a caller that must not wait +// (the XR pacing thread at a race exit). The device is still held while Dawn serializes. +void aurora_request_pipeline_cache_store(); // Allows the pipeline compiler to store the caches itself, rate-limited, whenever a first-use // burst completes. Off by default; enable it while a stall is acceptable, such as in menus. void aurora_set_pipeline_cache_idle_store(bool allowed); diff --git a/aurora-main/lib/aurora.cpp b/aurora-main/lib/aurora.cpp index 3c74e95..f5e4c35 100644 --- a/aurora-main/lib/aurora.cpp +++ b/aurora-main/lib/aurora.cpp @@ -2711,6 +2711,11 @@ void aurora_store_pipeline_caches() { aurora::gfx::store_pipeline_caches(); #endif } +void aurora_request_pipeline_cache_store() { +#ifdef AURORA_ENABLE_GX + aurora::gfx::request_pipeline_cache_store(); +#endif +} void aurora_set_pipeline_cache_idle_store(bool allowed) { #ifdef AURORA_ENABLE_GX aurora::gfx::set_pipeline_cache_idle_store(allowed); diff --git a/aurora-main/lib/gfx/pipeline_cache.cpp b/aurora-main/lib/gfx/pipeline_cache.cpp index 0038174..76bce45 100644 --- a/aurora-main/lib/gfx/pipeline_cache.cpp +++ b/aurora-main/lib/gfx/pipeline_cache.cpp @@ -1215,21 +1215,83 @@ static void stop_pipeline_cache_writer() { } void store_pipeline_caches() { - const uint64_t storesBefore = webgpu::blob_cache_stats().stores; + const auto before = webgpu::blob_cache_stats(); const auto start = std::chrono::steady_clock::now(); { std::lock_guard lock{g_storeMutex}; webgpu::serialize_pipeline_caches(); g_lastStore = std::chrono::steady_clock::now(); } + // Dawn holds its device only while it serializes; the blobs it handed over are compressed and + // committed by the cache's writer thread, and the recipes by the pipeline cache writer. + const auto serialized = std::chrono::steady_clock::now(); + webgpu::flush_cache_writes(); flush_pipeline_cache_writes(); - const uint64_t stored = webgpu::blob_cache_stats().stores - storesBefore; - if (stored != 0) { - const auto elapsed = std::chrono::duration_cast(std::chrono::steady_clock::now() - start); - Log.info("Stored the pipeline caches in {} ms: {} new Dawn blob cache entries", elapsed.count(), stored); + const auto after = webgpu::blob_cache_stats(); + const uint64_t stored = after.stores - before.stores; + const uint64_t unchanged = after.unchanged - before.unchanged; + if (stored != 0 || unchanged != 0) { + const auto milliseconds = [](auto duration) { + return std::chrono::duration_cast(duration).count(); + }; + Log.info("Stored the pipeline caches in {} ms, {} ms of it holding the GPU device: {} new Dawn blob cache " + "entries, {} unchanged", + milliseconds(std::chrono::steady_clock::now() - start), milliseconds(serialized - start), stored, + unchanged); } } +// A race exit asks for a store and moves on: the VR pacing thread that asks has to keep the +// compositor fed, while Dawn holds its device for as long as it serializes. +static std::mutex g_storeRequestMutex; +static std::condition_variable g_storeRequestCv; +static std::thread g_storeRequestThread; +static bool g_storeRequestsEnabled = false; +static bool g_storeRequested = false; +static bool g_storeRequestStop = false; + +static void store_request_loop() { + for (;;) { + { + std::unique_lock lock{g_storeRequestMutex}; + g_storeRequestCv.wait(lock, [] { return g_storeRequestStop || g_storeRequested; }); + if (!g_storeRequested) { + break; + } + g_storeRequested = false; + } + store_pipeline_caches(); + } +} + +void request_pipeline_cache_store() { + std::lock_guard lock{g_storeRequestMutex}; + if (!g_storeRequestsEnabled) { + return; + } + g_storeRequested = true; + if (!g_storeRequestThread.joinable()) { + g_storeRequestThread = std::thread(store_request_loop); + } + g_storeRequestCv.notify_one(); +} + +// A request still pending is carried out before this returns. +static void stop_store_requests() { + { + std::lock_guard lock{g_storeRequestMutex}; + g_storeRequestsEnabled = false; + g_storeRequestStop = true; + } + g_storeRequestCv.notify_all(); + if (g_storeRequestThread.joinable()) { + g_storeRequestThread.join(); + } + std::lock_guard lock{g_storeRequestMutex}; + g_storeRequested = false; + g_storeRequestStop = false; +} + void set_pipeline_cache_idle_store(bool allowed) noexcept { g_idleStoreAllowed.store(allowed, std::memory_order_relaxed); } @@ -1245,6 +1307,10 @@ PipelineRef find_pipeline(ShaderType type, const gx::PipelineConfig& config, New } void initialize_pipeline_cache() { + { + std::lock_guard lock{g_storeRequestMutex}; + g_storeRequestsEnabled = true; + } g_pipelineCacheBroken = false; g_pipelineCacheWriterStop = false; g_idleStoreAllowed.store(false, std::memory_order_relaxed); @@ -1274,6 +1340,8 @@ void initialize_pipeline_cache() { } void shutdown_pipeline_cache() { + // First: a requested store still needs the pipeline cache writer and the device. + stop_store_requests(); if (g_hasPipelineThread) { { std::lock_guard lock{g_pipelineMutex}; diff --git a/aurora-main/lib/gfx/pipeline_cache.hpp b/aurora-main/lib/gfx/pipeline_cache.hpp index c374f49..3d85b9e 100644 --- a/aurora-main/lib/gfx/pipeline_cache.hpp +++ b/aurora-main/lib/gfx/pipeline_cache.hpp @@ -23,10 +23,14 @@ void end_pipeline_frame(); void set_skip_unready_pipelines(bool enabled) noexcept; bool skip_unready_pipelines() noexcept; uint32_t queued_pipeline_count() noexcept; -// Persists the pipeline caches now: Dawn's Vulkan pipeline cache when a pipeline was compiled -// since the last store, then every queued recipe row. Holds the device for the store, so call it -// where a stall is invisible (a race exit, process exit). +// Persists the pipeline caches now: Dawn's Vulkan pipeline cache when a pipeline was created +// since the last store, then every queued recipe row, and returns once both are on disk. Dawn +// holds the device while it serializes its cache (tens to hundreds of ms for a large one), but +// not while the result is compressed and written. void store_pipeline_caches(); +// store_pipeline_caches on a background thread, returning at once. Requests made while one is +// running coalesce into one more store. +void request_pipeline_cache_store(); // Lets a drained first-use burst store the caches itself, rate-limited. Off by default; a host // turns it on while a stall is acceptable (menus) and off again for a race. void set_pipeline_cache_idle_store(bool allowed) noexcept; diff --git a/aurora-main/lib/webgpu/gpu.hpp b/aurora-main/lib/webgpu/gpu.hpp index eb2f1a4..23fcf2d 100644 --- a/aurora-main/lib/webgpu/gpu.hpp +++ b/aurora-main/lib/webgpu/gpu.hpp @@ -88,12 +88,17 @@ void draw_clear(const wgpu::RenderPassEncoder& pass, bool clearColor, bool clear size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valueSize, void* userdata); void store_to_cache(void const* key, size_t keySize, void const* value, size_t valueSize, void* userdata); void cache_shutdown(); +// Blocks until every blob Dawn has stored is committed to the cache database. Stores return +// once the blob is copied, so a caller that needs it on disk (a process about to end) waits here. +void flush_cache_writes(); struct BlobCacheStats { uint64_t lookups; uint64_t hits; uint64_t stores; uint64_t hitBytes; + // Stores of a large blob whose bytes matched what the database already holds. + uint64_t unchanged; }; BlobCacheStats blob_cache_stats() noexcept; // Persists the monolithic Vulkan pipeline cache into the blob cache (no-op elsewhere). diff --git a/aurora-main/lib/webgpu/gpu_cache.cpp b/aurora-main/lib/webgpu/gpu_cache.cpp index 4f52738..35e9642 100644 --- a/aurora-main/lib/webgpu/gpu_cache.cpp +++ b/aurora-main/lib/webgpu/gpu_cache.cpp @@ -1,9 +1,15 @@ #include +#include +#include #include #include +#include +#include #include #include #include +#include +#include #include #include "../fs_helper.hpp" @@ -28,9 +34,36 @@ static sqlite3_stmt* store_stmt; static sqlite3_stmt* touch_stmt; static bool cache_broken; static std::mutex cache_mutex; -#if defined(AURORA_CACHE_USE_ZSTD) -static std::vector compress_buffer; -#endif + +// Dawn calls store_to_cache while holding its device lock: the monolithic Vulkan pipeline cache is +// serialized under that lock and handed over as one blob (26 MiB on a desktop GPU, 58 MiB on a +// Quest 3), and compressing and committing it there held every other user of the device for the +// whole write, 0.4 s on a desktop and 0.9 to 3.8 s on a Quest at each race exit, long enough to +// stall the game thread and its music. The callback now only copies the blob and queues it, and +// one writer thread compresses and commits. Loads look at the queue first, so a stored blob can +// be read back at once. The newest content of each large blob is remembered by hash, so Dawn +// re-serializing an unchanged pipeline cache costs a hash instead of a rewrite. +struct KeyHasher { + size_t operator()(const XXH128_hash_t& hash) const noexcept { return static_cast(hash.low64 ^ hash.high64); } +}; +struct KeyEqual { + bool operator()(const XXH128_hash_t& a, const XXH128_hash_t& b) const noexcept { return XXH128_isEqual(a, b) != 0; } +}; +using QueuedBlob = std::shared_ptr>; + +constexpr size_t LargeBlobBytes = size_t{1} << 20; + +// Lock order: g_writeMutex and cache_mutex are never held together. +static std::mutex g_writeMutex; +static std::condition_variable g_writeCv; +// The newest content per key not yet committed. A key is in g_writeOrder at most once; a key in +// g_queuedBlobs but not in g_writeOrder is the one being written. +static std::unordered_map g_queuedBlobs; +static std::deque g_writeOrder; +static std::unordered_map g_largeContent; +static std::thread g_writerThread; +static bool g_writerStop = false; +static bool g_writerBusy = false; // Schema 3 added last_used (whole days since the Unix epoch) so stale blobs can be // pruned: config-version bumps and driver updates change every Dawn cache key, and @@ -47,6 +80,7 @@ static std::atomic g_lookups{0}; static std::atomic g_hits{0}; static std::atomic g_stores{0}; static std::atomic g_hitBytes{0}; +static std::atomic g_unchanged{0}; static std::vector g_pendingTouches; static int64_t days_now() { return static_cast(std::time(nullptr) / 86400); } @@ -259,9 +293,9 @@ static bool cache_init() { return true; } -size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valueSize, void*) { - std::lock_guard lock(cache_mutex); - +// Caller holds cache_mutex. `complete` is set when `value` received the whole entry. +static size_t load_from_database(const XXH128_hash_t& keyHash, void* value, size_t valueSize, bool& complete) { + complete = false; if (!cache_init()) { return 0; } @@ -272,13 +306,6 @@ size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valu return 0; } - // Dawn probes with value == nullptr for the size first, then fetches; count each - // probe as one logical lookup so the hit rate reads per-entry. - if (value == nullptr) { - g_lookups.fetch_add(1, std::memory_order_relaxed); - } - - const auto keyHash = XXH128(key, keySize, 0); check(sqlite3_bind_blob(load_stmt, 1, &keyHash, sizeof(keyHash), SQLITE_TRANSIENT)); const auto ret = sqlite3_step(load_stmt); @@ -321,6 +348,7 @@ size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valu std::memcpy(value, foundPtr, foundSize); } } + complete = foundSize != 0; } } else if (ret == SQLITE_DONE) { // Miss @@ -335,9 +363,77 @@ size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valu return foundSize; } -void store_to_cache(void const* key, size_t keySize, void const* value, size_t valueSize, void*) { - std::lock_guard lock(cache_mutex); +size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valueSize, void*) { + const auto keyHash = XXH128(key, keySize, 0); + // Dawn probes with value == nullptr for the size first, then fetches; count each + // probe as one logical lookup so the hit rate reads per-entry. + if (value == nullptr) { + g_lookups.fetch_add(1, std::memory_order_relaxed); + } + QueuedBlob queued; + { + std::lock_guard lock(g_writeMutex); + if (const auto entry = g_queuedBlobs.find(keyHash); entry != g_queuedBlobs.end()) { + queued = entry->second; + } + } + if (queued) { + if (value == nullptr) { + g_hits.fetch_add(1, std::memory_order_relaxed); + } else if (valueSize == queued->size()) { + std::memcpy(value, queued->data(), queued->size()); + g_hitBytes.fetch_add(queued->size(), std::memory_order_relaxed); + } + return queued->size(); + } + + bool complete = false; + size_t foundSize = 0; + { + std::lock_guard lock(cache_mutex); + foundSize = load_from_database(keyHash, value, valueSize, complete); + } + if (complete && foundSize >= LargeBlobBytes) { + // Remember what the database holds, so re-storing the same bytes is skipped. A store queued + // meanwhile already recorded newer content. + const auto content = XXH3_128bits(value, foundSize); + std::lock_guard lock(g_writeMutex); + g_largeContent.try_emplace(keyHash, content); + } + return foundSize; +} + +static void write_blob(const XXH128_hash_t& keyHash, const std::vector& blob, + std::vector& compressBuffer) { + const void* storedValue = blob.data(); + sqlite3_uint64 storedValueSize = blob.size(); + int compressed = 0; +#if defined(AURORA_CACHE_USE_ZSTD) + const auto bound = ZSTD_compressBound(blob.size()); + if (ZSTD_isError(bound)) { + Log.error("Failed to calculate ZSTD_compressBound: {}", ZSTD_getErrorName(bound)); + return; + } + + if (compressBuffer.size() < bound) { + compressBuffer.resize(bound); + } + + const auto compressRet = ZSTD_compress(compressBuffer.data(), compressBuffer.size(), blob.data(), blob.size(), 0); + if (ZSTD_isError(compressRet)) { + Log.error("ZSTD compression error: {}", ZSTD_getErrorName(compressRet)); + return; + } + + if (compressRet < blob.size()) { + storedValue = compressBuffer.data(); + storedValueSize = compressRet; + compressed = 1; + } +#endif + + std::lock_guard lock(cache_mutex); if (!cache_init()) { return; } @@ -348,61 +444,124 @@ void store_to_cache(void const* key, size_t keySize, void const* value, size_t v return; } - const void* storedValue = value; - sqlite3_uint64 storedValueSize = valueSize; - int compressed = 0; -#if defined(AURORA_CACHE_USE_ZSTD) - const auto bound = ZSTD_compressBound(valueSize); - if (ZSTD_isError(bound)) { - Log.error("Failed to calculate ZSTD_compressBound: {}", ZSTD_getErrorName(bound)); - return; - } - - if (compress_buffer.size() < bound) { - compress_buffer.resize(bound); - } - - const auto compressRet = ZSTD_compress(compress_buffer.data(), compress_buffer.size(), value, valueSize, 0); - if (ZSTD_isError(compressRet)) { - Log.error("ZSTD compression error: {}", ZSTD_getErrorName(compressRet)); - return; - } - - if (compressRet < valueSize) { - storedValue = compress_buffer.data(); - storedValueSize = compressRet; - compressed = 1; - } -#endif - - const auto keyHash = XXH128(key, keySize, 0); + // Both buffers outlive the statement's use of them: the binding is cleared below. check(sqlite3_bind_blob64(store_stmt, 1, &keyHash, sizeof(keyHash), SQLITE_TRANSIENT)); - check( - sqlite3_bind_blob64(store_stmt, 2, storedValue, storedValueSize, compressed ? SQLITE_STATIC : SQLITE_TRANSIENT)); - check(sqlite3_bind_int64(store_stmt, 3, static_cast(valueSize))); + check(sqlite3_bind_blob64(store_stmt, 2, storedValue, storedValueSize, SQLITE_STATIC)); + check(sqlite3_bind_int64(store_stmt, 3, static_cast(blob.size()))); check(sqlite3_bind_int(store_stmt, 4, compressed)); check(sqlite3_bind_int64(store_stmt, 5, days_now())); - g_stores.fetch_add(1, std::memory_order_relaxed); const auto ret = sqlite3_step(store_stmt); + check(sqlite3_reset(store_stmt)); + check(sqlite3_bind_null(store_stmt, 2)); if (ret != SQLITE_DONE) { - // Error or something Log.error("Failed to insert row: {}", sqlite3_errmsg(db)); return; } - check(sqlite3_reset(store_stmt)); - check(sqlite3_bind_null(store_stmt, 2)); - check(sqlite3_bind_null(store_stmt, 4)); - tx.commit(); } +static void cache_writer_loop() { + std::vector compressBuffer; + for (;;) { + XXH128_hash_t keyHash{}; + QueuedBlob blob; + { + std::unique_lock lock(g_writeMutex); + g_writeCv.wait(lock, [] { return g_writerStop || !g_writeOrder.empty(); }); + if (g_writeOrder.empty()) { + // Stopping, and everything queued has been written. + break; + } + keyHash = g_writeOrder.front(); + g_writeOrder.pop_front(); + blob = g_queuedBlobs.at(keyHash); + g_writerBusy = true; + } + + const auto start = std::chrono::steady_clock::now(); + write_blob(keyHash, *blob, compressBuffer); + if (blob->size() >= LargeBlobBytes) { + const auto elapsed = + std::chrono::duration_cast(std::chrono::steady_clock::now() - start); + Log.info("Wrote a {:.1f} MiB Dawn cache blob in {} ms, off the GPU device", + static_cast(blob->size()) / (1024.0 * 1024.0), elapsed.count()); + } + + { + std::lock_guard lock(g_writeMutex); + if (const auto entry = g_queuedBlobs.find(keyHash); entry != g_queuedBlobs.end()) { + if (entry->second == blob) { + g_queuedBlobs.erase(entry); + } else { + // Stored again while this copy was being written: the newer content still goes out. + g_writeOrder.push_back(keyHash); + } + } + g_writerBusy = false; + } + g_writeCv.notify_all(); + } +} + +void store_to_cache(void const* key, size_t keySize, void const* value, size_t valueSize, void*) { + const auto keyHash = XXH128(key, keySize, 0); + const auto* bytes = static_cast(value); + const bool large = valueSize >= LargeBlobBytes; + const XXH128_hash_t content = large ? XXH3_128bits(value, valueSize) : XXH128_hash_t{}; + if (large) { + std::lock_guard lock(g_writeMutex); + if (const auto known = g_largeContent.find(keyHash); + known != g_largeContent.end() && XXH128_isEqual(known->second, content)) { + g_unchanged.fetch_add(1, std::memory_order_relaxed); + return; + } + } + + // Dawn's buffer is only valid for this call. + auto blob = std::make_shared>(bytes, bytes + valueSize); + { + std::lock_guard lock(g_writeMutex); + if (large) { + g_largeContent.insert_or_assign(keyHash, content); + } + if (g_queuedBlobs.insert_or_assign(keyHash, std::move(blob)).second) { + g_writeOrder.push_back(keyHash); + } + if (!g_writerThread.joinable()) { + g_writerStop = false; + g_writerThread = std::thread(cache_writer_loop); + } + } + g_stores.fetch_add(1, std::memory_order_relaxed); + g_writeCv.notify_all(); +} + +void flush_cache_writes() { + std::unique_lock lock(g_writeMutex); + g_writeCv.wait(lock, [] { return g_writeOrder.empty() && !g_writerBusy; }); +} + void cache_shutdown() { + { + std::lock_guard lock(g_writeMutex); + g_writerStop = true; + } + g_writeCv.notify_all(); + // The writer drains the queue before it exits. + if (g_writerThread.joinable()) { + g_writerThread.join(); + } + { + std::lock_guard lock(g_writeMutex); + g_queuedBlobs.clear(); + g_writeOrder.clear(); + g_largeContent.clear(); + g_writerStop = false; + } + std::lock_guard lock(cache_mutex); -#if defined(AURORA_CACHE_USE_ZSTD) - compress_buffer.clear(); -#endif flush_touches(); check(sqlite3_finalize(load_stmt)); check(sqlite3_finalize(store_stmt)); @@ -420,6 +579,7 @@ BlobCacheStats blob_cache_stats() noexcept { .hits = g_hits.load(std::memory_order_relaxed), .stores = g_stores.load(std::memory_order_relaxed), .hitBytes = g_hitBytes.load(std::memory_order_relaxed), + .unchanged = g_unchanged.load(std::memory_order_relaxed), }; } diff --git a/aurora-main/tests/CMakeLists.txt b/aurora-main/tests/CMakeLists.txt index fdc8147..3633746 100644 --- a/aurora-main/tests/CMakeLists.txt +++ b/aurora-main/tests/CMakeLists.txt @@ -106,6 +106,20 @@ if (AURORA_ENABLE_GX) ) gtest_discover_tests(gx_fifo_tests) + + # The Dawn blob cache's background writer, against a real SQLite database in a temp folder. + if (TARGET sqlite3) + add_executable(gpu_cache_tests gpu_cache_test.cpp ../lib/webgpu/gpu_cache.cpp) + target_include_directories(gpu_cache_tests PRIVATE ../include ../lib) + target_compile_definitions(gpu_cache_tests PRIVATE AURORA TARGET_PC) + target_link_libraries(gpu_cache_tests PRIVATE gtest gtest_main fmt::fmt xxhash sqlite3 dawn::dawncpp_headers + TracyClient ${AURORA_SDL3_TARGET}) + if (AURORA_CACHE_USE_ZSTD) + target_compile_definitions(gpu_cache_tests PRIVATE AURORA_CACHE_USE_ZSTD) + target_link_libraries(gpu_cache_tests PRIVATE libzstd_static) + endif () + gtest_discover_tests(gpu_cache_tests) + endif () endif () # AURORA_ENABLE_GX # DVD API tests diff --git a/aurora-main/tests/gpu_cache_test.cpp b/aurora-main/tests/gpu_cache_test.cpp new file mode 100644 index 0000000..8912c60 --- /dev/null +++ b/aurora-main/tests/gpu_cache_test.cpp @@ -0,0 +1,159 @@ +// The Dawn blob cache writes off Dawn's device lock: store_to_cache copies and queues, a writer +// thread compresses and commits. These tests pin what callers rely on: a stored blob reads back +// at once, reaches the database, is not rewritten when unchanged, and survives shutdown. + +#include "webgpu/gpu.hpp" + +#include "internal.hpp" + +#include + +#include +#include +#include +#include +#include + +namespace aurora { +AuroraConfig g_config{}; +void log_internal(AuroraLogLevel, const char*, const char* message, unsigned int len) noexcept { + std::fprintf(stderr, "%.*s\n", static_cast(len), message); +} +void Module::show_fatal_dialog(const char*, std::string_view) noexcept {} +} // namespace aurora + +auto fmt::formatter::format(AuroraLogLevel level, format_context& ctx) const + -> format_context::iterator { + return fmt::format_to(ctx.out(), "{}", static_cast(level)); +} + +namespace { + +using aurora::webgpu::blob_cache_stats; +using aurora::webgpu::cache_shutdown; +using aurora::webgpu::flush_cache_writes; +using aurora::webgpu::load_from_cache; +using aurora::webgpu::store_to_cache; + +class GpuCacheTest : public ::testing::Test { +protected: + void SetUp() override { + m_directory = std::filesystem::temp_directory_path() / + ("aurora_gpu_cache_test_" + std::string(::testing::UnitTest::GetInstance()->current_test_info()->name())); + std::filesystem::remove_all(m_directory); + std::filesystem::create_directories(m_directory); + m_path = m_directory.string(); + aurora::g_config.cachePath = m_path.c_str(); + } + + void TearDown() override { + cache_shutdown(); + std::filesystem::remove_all(m_directory); + } + + static void Store(const std::string& key, const std::vector& value) { + store_to_cache(key.data(), key.size(), value.data(), value.size(), nullptr); + } + + // Dawn's protocol: probe for the size, then fetch into a buffer of that size. + static std::vector Load(const std::string& key) { + const size_t size = load_from_cache(key.data(), key.size(), nullptr, 0, nullptr); + std::vector value(size); + if (size != 0) { + EXPECT_EQ(load_from_cache(key.data(), key.size(), value.data(), value.size(), nullptr), size); + } + return value; + } + + // Pseudo-random bytes, so compression cannot hide a mix-up between two blobs. + static std::vector Bytes(size_t size, uint32_t seed) { + std::vector value(size); + uint32_t state = seed * 2654435761u + 1u; + for (auto& byte : value) { + state = state * 1664525u + 1013904223u; + byte = static_cast(state >> 24); + } + return value; + } + + std::filesystem::path m_directory; + std::string m_path; +}; + +TEST_F(GpuCacheTest, StoredBlobReadsBackBeforeAndAfterItIsCommitted) { + const auto value = Bytes(4096, 1); + Store("pipeline", value); + EXPECT_EQ(Load("pipeline"), value); + flush_cache_writes(); + EXPECT_EQ(Load("pipeline"), value); + EXPECT_TRUE(Load("missing").empty()); +} + +TEST_F(GpuCacheTest, ShutdownWritesWhatIsStillQueued) { + const auto value = Bytes(3 << 20, 2); + Store("monolithic", value); + cache_shutdown(); + // The next call reopens the database: only a committed row can answer now. + EXPECT_EQ(Load("monolithic"), value); +} + +TEST_F(GpuCacheTest, NewestStoreOfAKeyWins) { + const auto first = Bytes(2 << 20, 3); + const auto second = Bytes(2 << 20, 4); + Store("monolithic", first); + Store("monolithic", second); + EXPECT_EQ(Load("monolithic"), second); + flush_cache_writes(); + cache_shutdown(); + EXPECT_EQ(Load("monolithic"), second); +} + +TEST_F(GpuCacheTest, UnchangedLargeBlobIsNotRewritten) { + const auto value = Bytes(2 << 20, 5); + Store("monolithic", value); + flush_cache_writes(); + const auto before = blob_cache_stats(); + Store("monolithic", value); + auto after = blob_cache_stats(); + EXPECT_EQ(after.stores, before.stores); + EXPECT_EQ(after.unchanged, before.unchanged + 1); + + auto changed = value; + changed[changed.size() / 2] ^= 0xFF; + Store("monolithic", changed); + after = blob_cache_stats(); + EXPECT_EQ(after.stores, before.stores + 1); + EXPECT_EQ(Load("monolithic"), changed); + + // Changing back is a change too. + Store("monolithic", value); + EXPECT_EQ(blob_cache_stats().stores, before.stores + 2); + EXPECT_EQ(Load("monolithic"), value); +} + +TEST_F(GpuCacheTest, BlobLoadedAtStartupIsNotRewrittenUnchanged) { + const auto value = Bytes(2 << 20, 6); + Store("monolithic", value); + cache_shutdown(); + + // A new session loads the blob Dawn saved last time, then serializes the same bytes again. + ASSERT_EQ(Load("monolithic"), value); + const auto before = blob_cache_stats(); + Store("monolithic", value); + const auto after = blob_cache_stats(); + EXPECT_EQ(after.stores, before.stores); + EXPECT_EQ(after.unchanged, before.unchanged + 1); +} + +TEST_F(GpuCacheTest, SmallBlobsAreAlwaysWritten) { + const auto value = Bytes(512, 7); + Store("shader", value); + flush_cache_writes(); + const auto before = blob_cache_stats(); + Store("shader", value); + EXPECT_EQ(blob_cache_stats().stores, before.stores + 1); + flush_cache_writes(); + EXPECT_EQ(Load("shader"), value); +} + +} // namespace diff --git a/docs/quest-port.md b/docs/quest-port.md index 0138850..e1ab9e2 100644 --- a/docs/quest-port.md +++ b/docs/quest-port.md @@ -603,6 +603,7 @@ Bring-up fixes that only a device could reveal: | Crypto++ `cpu-features.h` not found | The NDK ships cpu-features as source | Compiled into `mkw_cryptopp` on Android | | Exploded racers and menu characters; smeared movie panels in the menus; then, once those were fixed, damaged eyes and slightly misplaced detail on characters | The Adreno 740 driver reads the wrong bytes when the shader multiplies an index by a stride that is not a multiple of 4. That covers the vertex fetch (`ubuf.vtx_start + vidx * stride + offset`) and indexed array reads (`array_start + index * stride`, e.g. 6-byte S16 normals). GX packs both byte-tight, so skinned models (a 1-byte `PNMTXIDX` first, stride 7) broke everywhere | Android pads every uploaded vertex and every indexed-array element to a 4-byte stride (`padded_upload_stride` in `lib/gx/gx.cpp`). Offsets inside a vertex or element are unchanged, and desktop is unchanged. **Fixed, headset-verified 2026-09-16** at character select and a Grand Prix start | | Every launch recompiled every shader: a 14 to 34 s prewarm, and the Dawn blob cache reporting exactly one miss and no stores | Dawn's monolithic Vulkan pipeline cache is only written by `PerformIdleTasks`, which `gpu.cpp` resolved through the Windows DLL alone, and the quit path ends the process without `aurora_shutdown`, so nothing compiled after prewarm was kept either | Static Dawn calls it directly; `aurora_store_pipeline_caches` runs at a race exit, when the session loses focus and on the quit path, and the compiler stores idle bursts itself while the headset shows the virtual screen. The unpacked `initial_pipeline_cache.db` is also refreshed per APK install now | +| The game and its music froze for 0.9 to 3.8 s at the end of a race, longer as a session went on | Dawn serializes its monolithic Vulkan pipeline cache (58 MiB on the Quest, and it barely compresses) under its device lock and hands it to Aurora's blob-cache callback, which compressed it and committed it to SQLite still inside that lock; the frame worker waited on the device and the game thread on the frame worker. The store also ran on the XR pacing thread | `gpu_cache.cpp` copies the blob and queues it; a writer thread compresses and commits, and bytes identical to what the database holds are not rewritten. The race exit uses `aurora_request_pipeline_cache_store`, off the pacing thread. The log's `Stored the pipeline caches` line reports the part still spent holding the device | How the explosion was isolated, so the next Adreno rendering bug starts further ahead: diff --git a/runtime/src/vr/openxr_integration.cpp b/runtime/src/vr/openxr_integration.cpp index 02948ed..1224ff5 100644 --- a/runtime/src/vr/openxr_integration.cpp +++ b/runtime/src/vr/openxr_integration.cpp @@ -785,6 +785,7 @@ private: ResetTrackingOrigin(); // Nothing is displayed while the session is not running (the system menu, // the headset taken off), so the stall of a cache store is invisible here. + // Waiting for it is deliberate: the process may be ended next. aurora_store_pipeline_caches(); } } @@ -851,9 +852,11 @@ private: aurora_set_stereo_panel_layer(panel_layer); presentation.panel.requested = panel_layer && OpenXRSettingsPanelOpen(); - // Pipeline caches are stored where their stall is invisible: once when a race ends, - // and by the compiler itself while the headset shows the virtual screen. Never - // mid-race, and a race on the virtual screen (Flat Screen mode) is still a race. + // Pipeline caches are stored where their stall is least visible: once when a race + // ends, and by the compiler itself while the headset shows the virtual screen. Never + // mid-race, and a race on the virtual screen (Flat Screen mode) is still a race. The + // race-exit store runs on Aurora's thread: this one keeps submitting frames while + // Dawn holds its device to serialize, which is still a brief game stall. const bool racing = immersive || (!policy.config.immersive_races && policy.scene.mode == VRSceneMode::Race); if (!store_gate_set || racing != store_gate_racing) { @@ -862,7 +865,7 @@ private: store_gate_racing = racing; aurora_set_pipeline_cache_idle_store(!racing); if (left_race) { - aurora_store_pipeline_caches(); + aurora_request_pipeline_cache_store(); } }