Implement background pipeline cache storage

This commit is contained in:
iChris4 committed 2026-09-23 23:53:51 +02:00
1 parent cb2b439f6a
commit 1a4b0589f4
10 files changed
+492 -69

No files matched your search

+6 -2
View File
@@ -334,9 +334,13 @@ bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros);
// the gap between aurora_end_frame() and aurora_begin_frame(). // the gap between aurora_end_frame() and aurora_begin_frame().
void aurora_quiesce_frame_worker(); void aurora_quiesce_frame_worker();
// Persists the renderer's pipeline caches now (Dawn's Vulkan pipeline cache, then the queued // Persists the renderer's pipeline caches now (Dawn's Vulkan pipeline cache, then the queued
// pipeline recipes). It holds the GPU device for the store, so a race would see it as a stall: // pipeline recipes) and returns once they are on disk. Dawn holds the GPU device while it
// call it at a race exit, when the session loses focus, or before ending the process. // serializes its cache, so a race would see that part as a stall: call it at a race exit, when
// the session loses focus, or before ending the process.
void aurora_store_pipeline_caches(); void aurora_store_pipeline_caches();
// The same store on a background thread, returning at once, for a caller that must not wait
// (the XR pacing thread at a race exit). The device is still held while Dawn serializes.
void aurora_request_pipeline_cache_store();
// Allows the pipeline compiler to store the caches itself, rate-limited, whenever a first-use // Allows the pipeline compiler to store the caches itself, rate-limited, whenever a first-use
// burst completes. Off by default; enable it while a stall is acceptable, such as in menus. // burst completes. Off by default; enable it while a stall is acceptable, such as in menus.
void aurora_set_pipeline_cache_idle_store(bool allowed); void aurora_set_pipeline_cache_idle_store(bool allowed);
+5
View File
@@ -2711,6 +2711,11 @@ void aurora_store_pipeline_caches() {
aurora::gfx::store_pipeline_caches(); aurora::gfx::store_pipeline_caches();
#endif #endif
} }
void aurora_request_pipeline_cache_store() {
#ifdef AURORA_ENABLE_GX
aurora::gfx::request_pipeline_cache_store();
#endif
}
void aurora_set_pipeline_cache_idle_store(bool allowed) { void aurora_set_pipeline_cache_idle_store(bool allowed) {
#ifdef AURORA_ENABLE_GX #ifdef AURORA_ENABLE_GX
aurora::gfx::set_pipeline_cache_idle_store(allowed); aurora::gfx::set_pipeline_cache_idle_store(allowed);
+73 -5
View File
@@ -1215,21 +1215,83 @@ static void stop_pipeline_cache_writer() {
} }
void store_pipeline_caches() { void store_pipeline_caches() {
const uint64_t storesBefore = webgpu::blob_cache_stats().stores; const auto before = webgpu::blob_cache_stats();
const auto start = std::chrono::steady_clock::now(); const auto start = std::chrono::steady_clock::now();
{ {
std::lock_guard lock{g_storeMutex}; std::lock_guard lock{g_storeMutex};
webgpu::serialize_pipeline_caches(); webgpu::serialize_pipeline_caches();
g_lastStore = std::chrono::steady_clock::now(); g_lastStore = std::chrono::steady_clock::now();
} }
// Dawn holds its device only while it serializes; the blobs it handed over are compressed and
// committed by the cache's writer thread, and the recipes by the pipeline cache writer.
const auto serialized = std::chrono::steady_clock::now();
webgpu::flush_cache_writes();
flush_pipeline_cache_writes(); flush_pipeline_cache_writes();
const uint64_t stored = webgpu::blob_cache_stats().stores - storesBefore; const auto after = webgpu::blob_cache_stats();
if (stored != 0) { const uint64_t stored = after.stores - before.stores;
const auto elapsed = std::chrono::duration_cast<std::chrono::milliseconds>(std::chrono::steady_clock::now() - start); const uint64_t unchanged = after.unchanged - before.unchanged;
Log.info("Stored the pipeline caches in {} ms: {} new Dawn blob cache entries", elapsed.count(), stored); if (stored != 0 || unchanged != 0) {
const auto milliseconds = [](auto duration) {
return std::chrono::duration_cast<std::chrono::milliseconds>(duration).count();
};
Log.info("Stored the pipeline caches in {} ms, {} ms of it holding the GPU device: {} new Dawn blob cache "
"entries, {} unchanged",
milliseconds(std::chrono::steady_clock::now() - start), milliseconds(serialized - start), stored,
unchanged);
} }
} }
// A race exit asks for a store and moves on: the VR pacing thread that asks has to keep the
// compositor fed, while Dawn holds its device for as long as it serializes.
static std::mutex g_storeRequestMutex;
static std::condition_variable g_storeRequestCv;
static std::thread g_storeRequestThread;
static bool g_storeRequestsEnabled = false;
static bool g_storeRequested = false;
static bool g_storeRequestStop = false;
static void store_request_loop() {
for (;;) {
{
std::unique_lock lock{g_storeRequestMutex};
g_storeRequestCv.wait(lock, [] { return g_storeRequestStop || g_storeRequested; });
if (!g_storeRequested) {
break;
}
g_storeRequested = false;
}
store_pipeline_caches();
}
}
void request_pipeline_cache_store() {
std::lock_guard lock{g_storeRequestMutex};
if (!g_storeRequestsEnabled) {
return;
}
g_storeRequested = true;
if (!g_storeRequestThread.joinable()) {
g_storeRequestThread = std::thread(store_request_loop);
}
g_storeRequestCv.notify_one();
}
// A request still pending is carried out before this returns.
static void stop_store_requests() {
{
std::lock_guard lock{g_storeRequestMutex};
g_storeRequestsEnabled = false;
g_storeRequestStop = true;
}
g_storeRequestCv.notify_all();
if (g_storeRequestThread.joinable()) {
g_storeRequestThread.join();
}
std::lock_guard lock{g_storeRequestMutex};
g_storeRequested = false;
g_storeRequestStop = false;
}
void set_pipeline_cache_idle_store(bool allowed) noexcept { void set_pipeline_cache_idle_store(bool allowed) noexcept {
g_idleStoreAllowed.store(allowed, std::memory_order_relaxed); g_idleStoreAllowed.store(allowed, std::memory_order_relaxed);
} }
@@ -1245,6 +1307,10 @@ PipelineRef find_pipeline(ShaderType type, const gx::PipelineConfig& config, New
} }
void initialize_pipeline_cache() { void initialize_pipeline_cache() {
{
std::lock_guard lock{g_storeRequestMutex};
g_storeRequestsEnabled = true;
}
g_pipelineCacheBroken = false; g_pipelineCacheBroken = false;
g_pipelineCacheWriterStop = false; g_pipelineCacheWriterStop = false;
g_idleStoreAllowed.store(false, std::memory_order_relaxed); g_idleStoreAllowed.store(false, std::memory_order_relaxed);
@@ -1274,6 +1340,8 @@ void initialize_pipeline_cache() {
} }
void shutdown_pipeline_cache() { void shutdown_pipeline_cache() {
// First: a requested store still needs the pipeline cache writer and the device.
stop_store_requests();
if (g_hasPipelineThread) { if (g_hasPipelineThread) {
{ {
std::lock_guard lock{g_pipelineMutex}; std::lock_guard lock{g_pipelineMutex};
+7 -3
View File
@@ -23,10 +23,14 @@ void end_pipeline_frame();
void set_skip_unready_pipelines(bool enabled) noexcept; void set_skip_unready_pipelines(bool enabled) noexcept;
bool skip_unready_pipelines() noexcept; bool skip_unready_pipelines() noexcept;
uint32_t queued_pipeline_count() noexcept; uint32_t queued_pipeline_count() noexcept;
// Persists the pipeline caches now: Dawn's Vulkan pipeline cache when a pipeline was compiled // Persists the pipeline caches now: Dawn's Vulkan pipeline cache when a pipeline was created
// since the last store, then every queued recipe row. Holds the device for the store, so call it // since the last store, then every queued recipe row, and returns once both are on disk. Dawn
// where a stall is invisible (a race exit, process exit). // holds the device while it serializes its cache (tens to hundreds of ms for a large one), but
// not while the result is compressed and written.
void store_pipeline_caches(); void store_pipeline_caches();
// store_pipeline_caches on a background thread, returning at once. Requests made while one is
// running coalesce into one more store.
void request_pipeline_cache_store();
// Lets a drained first-use burst store the caches itself, rate-limited. Off by default; a host // Lets a drained first-use burst store the caches itself, rate-limited. Off by default; a host
// turns it on while a stall is acceptable (menus) and off again for a race. // turns it on while a stall is acceptable (menus) and off again for a race.
void set_pipeline_cache_idle_store(bool allowed) noexcept; void set_pipeline_cache_idle_store(bool allowed) noexcept;
+5
View File
@@ -88,12 +88,17 @@ void draw_clear(const wgpu::RenderPassEncoder& pass, bool clearColor, bool clear
size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valueSize, void* userdata); size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valueSize, void* userdata);
void store_to_cache(void const* key, size_t keySize, void const* value, size_t valueSize, void* userdata); void store_to_cache(void const* key, size_t keySize, void const* value, size_t valueSize, void* userdata);
void cache_shutdown(); void cache_shutdown();
// Blocks until every blob Dawn has stored is committed to the cache database. Stores return
// once the blob is copied, so a caller that needs it on disk (a process about to end) waits here.
void flush_cache_writes();
struct BlobCacheStats { struct BlobCacheStats {
uint64_t lookups; uint64_t lookups;
uint64_t hits; uint64_t hits;
uint64_t stores; uint64_t stores;
uint64_t hitBytes; uint64_t hitBytes;
// Stores of a large blob whose bytes matched what the database already holds.
uint64_t unchanged;
}; };
BlobCacheStats blob_cache_stats() noexcept; BlobCacheStats blob_cache_stats() noexcept;
// Persists the monolithic Vulkan pipeline cache into the blob cache (no-op elsewhere). // Persists the monolithic Vulkan pipeline cache into the blob cache (no-op elsewhere).
+215 -55
View File
@@ -1,9 +1,15 @@
#include <atomic> #include <atomic>
#include <chrono>
#include <condition_variable>
#include <cstring> #include <cstring>
#include <ctime> #include <ctime>
#include <deque>
#include <memory>
#include <mutex> #include <mutex>
#include <string> #include <string>
#include <filesystem> #include <filesystem>
#include <thread>
#include <unordered_map>
#include <vector> #include <vector>
#include "../fs_helper.hpp" #include "../fs_helper.hpp"
@@ -28,9 +34,36 @@ static sqlite3_stmt* store_stmt;
static sqlite3_stmt* touch_stmt; static sqlite3_stmt* touch_stmt;
static bool cache_broken; static bool cache_broken;
static std::mutex cache_mutex; static std::mutex cache_mutex;
#if defined(AURORA_CACHE_USE_ZSTD)
static std::vector<uint8_t> compress_buffer; // Dawn calls store_to_cache while holding its device lock: the monolithic Vulkan pipeline cache is
#endif // serialized under that lock and handed over as one blob (26 MiB on a desktop GPU, 58 MiB on a
// Quest 3), and compressing and committing it there held every other user of the device for the
// whole write, 0.4 s on a desktop and 0.9 to 3.8 s on a Quest at each race exit, long enough to
// stall the game thread and its music. The callback now only copies the blob and queues it, and
// one writer thread compresses and commits. Loads look at the queue first, so a stored blob can
// be read back at once. The newest content of each large blob is remembered by hash, so Dawn
// re-serializing an unchanged pipeline cache costs a hash instead of a rewrite.
struct KeyHasher {
size_t operator()(const XXH128_hash_t& hash) const noexcept { return static_cast<size_t>(hash.low64 ^ hash.high64); }
};
struct KeyEqual {
bool operator()(const XXH128_hash_t& a, const XXH128_hash_t& b) const noexcept { return XXH128_isEqual(a, b) != 0; }
};
using QueuedBlob = std::shared_ptr<const std::vector<uint8_t>>;
constexpr size_t LargeBlobBytes = size_t{1} << 20;
// Lock order: g_writeMutex and cache_mutex are never held together.
static std::mutex g_writeMutex;
static std::condition_variable g_writeCv;
// The newest content per key not yet committed. A key is in g_writeOrder at most once; a key in
// g_queuedBlobs but not in g_writeOrder is the one being written.
static std::unordered_map<XXH128_hash_t, QueuedBlob, KeyHasher, KeyEqual> g_queuedBlobs;
static std::deque<XXH128_hash_t> g_writeOrder;
static std::unordered_map<XXH128_hash_t, XXH128_hash_t, KeyHasher, KeyEqual> g_largeContent;
static std::thread g_writerThread;
static bool g_writerStop = false;
static bool g_writerBusy = false;
// Schema 3 added last_used (whole days since the Unix epoch) so stale blobs can be // Schema 3 added last_used (whole days since the Unix epoch) so stale blobs can be
// pruned: config-version bumps and driver updates change every Dawn cache key, and // pruned: config-version bumps and driver updates change every Dawn cache key, and
@@ -47,6 +80,7 @@ static std::atomic<uint64_t> g_lookups{0};
static std::atomic<uint64_t> g_hits{0}; static std::atomic<uint64_t> g_hits{0};
static std::atomic<uint64_t> g_stores{0}; static std::atomic<uint64_t> g_stores{0};
static std::atomic<uint64_t> g_hitBytes{0}; static std::atomic<uint64_t> g_hitBytes{0};
static std::atomic<uint64_t> g_unchanged{0};
static std::vector<XXH128_hash_t> g_pendingTouches; static std::vector<XXH128_hash_t> g_pendingTouches;
static int64_t days_now() { return static_cast<int64_t>(std::time(nullptr) / 86400); } static int64_t days_now() { return static_cast<int64_t>(std::time(nullptr) / 86400); }
@@ -259,9 +293,9 @@ static bool cache_init() {
return true; return true;
} }
size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valueSize, void*) { // Caller holds cache_mutex. `complete` is set when `value` received the whole entry.
std::lock_guard lock(cache_mutex); static size_t load_from_database(const XXH128_hash_t& keyHash, void* value, size_t valueSize, bool& complete) {
complete = false;
if (!cache_init()) { if (!cache_init()) {
return 0; return 0;
} }
@@ -272,13 +306,6 @@ size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valu
return 0; return 0;
} }
// Dawn probes with value == nullptr for the size first, then fetches; count each
// probe as one logical lookup so the hit rate reads per-entry.
if (value == nullptr) {
g_lookups.fetch_add(1, std::memory_order_relaxed);
}
const auto keyHash = XXH128(key, keySize, 0);
check(sqlite3_bind_blob(load_stmt, 1, &keyHash, sizeof(keyHash), SQLITE_TRANSIENT)); check(sqlite3_bind_blob(load_stmt, 1, &keyHash, sizeof(keyHash), SQLITE_TRANSIENT));
const auto ret = sqlite3_step(load_stmt); const auto ret = sqlite3_step(load_stmt);
@@ -321,6 +348,7 @@ size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valu
std::memcpy(value, foundPtr, foundSize); std::memcpy(value, foundPtr, foundSize);
} }
} }
complete = foundSize != 0;
} }
} else if (ret == SQLITE_DONE) { } else if (ret == SQLITE_DONE) {
// Miss // Miss
@@ -335,9 +363,77 @@ size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valu
return foundSize; return foundSize;
} }
void store_to_cache(void const* key, size_t keySize, void const* value, size_t valueSize, void*) { size_t load_from_cache(void const* key, size_t keySize, void* value, size_t valueSize, void*) {
std::lock_guard lock(cache_mutex); const auto keyHash = XXH128(key, keySize, 0);
// Dawn probes with value == nullptr for the size first, then fetches; count each
// probe as one logical lookup so the hit rate reads per-entry.
if (value == nullptr) {
g_lookups.fetch_add(1, std::memory_order_relaxed);
}
QueuedBlob queued;
{
std::lock_guard lock(g_writeMutex);
if (const auto entry = g_queuedBlobs.find(keyHash); entry != g_queuedBlobs.end()) {
queued = entry->second;
}
}
if (queued) {
if (value == nullptr) {
g_hits.fetch_add(1, std::memory_order_relaxed);
} else if (valueSize == queued->size()) {
std::memcpy(value, queued->data(), queued->size());
g_hitBytes.fetch_add(queued->size(), std::memory_order_relaxed);
}
return queued->size();
}
bool complete = false;
size_t foundSize = 0;
{
std::lock_guard lock(cache_mutex);
foundSize = load_from_database(keyHash, value, valueSize, complete);
}
if (complete && foundSize >= LargeBlobBytes) {
// Remember what the database holds, so re-storing the same bytes is skipped. A store queued
// meanwhile already recorded newer content.
const auto content = XXH3_128bits(value, foundSize);
std::lock_guard lock(g_writeMutex);
g_largeContent.try_emplace(keyHash, content);
}
return foundSize;
}
static void write_blob(const XXH128_hash_t& keyHash, const std::vector<uint8_t>& blob,
std::vector<uint8_t>& compressBuffer) {
const void* storedValue = blob.data();
sqlite3_uint64 storedValueSize = blob.size();
int compressed = 0;
#if defined(AURORA_CACHE_USE_ZSTD)
const auto bound = ZSTD_compressBound(blob.size());
if (ZSTD_isError(bound)) {
Log.error("Failed to calculate ZSTD_compressBound: {}", ZSTD_getErrorName(bound));
return;
}
if (compressBuffer.size() < bound) {
compressBuffer.resize(bound);
}
const auto compressRet = ZSTD_compress(compressBuffer.data(), compressBuffer.size(), blob.data(), blob.size(), 0);
if (ZSTD_isError(compressRet)) {
Log.error("ZSTD compression error: {}", ZSTD_getErrorName(compressRet));
return;
}
if (compressRet < blob.size()) {
storedValue = compressBuffer.data();
storedValueSize = compressRet;
compressed = 1;
}
#endif
std::lock_guard lock(cache_mutex);
if (!cache_init()) { if (!cache_init()) {
return; return;
} }
@@ -348,61 +444,124 @@ void store_to_cache(void const* key, size_t keySize, void const* value, size_t v
return; return;
} }
const void* storedValue = value; // Both buffers outlive the statement's use of them: the binding is cleared below.
sqlite3_uint64 storedValueSize = valueSize;
int compressed = 0;
#if defined(AURORA_CACHE_USE_ZSTD)
const auto bound = ZSTD_compressBound(valueSize);
if (ZSTD_isError(bound)) {
Log.error("Failed to calculate ZSTD_compressBound: {}", ZSTD_getErrorName(bound));
return;
}
if (compress_buffer.size() < bound) {
compress_buffer.resize(bound);
}
const auto compressRet = ZSTD_compress(compress_buffer.data(), compress_buffer.size(), value, valueSize, 0);
if (ZSTD_isError(compressRet)) {
Log.error("ZSTD compression error: {}", ZSTD_getErrorName(compressRet));
return;
}
if (compressRet < valueSize) {
storedValue = compress_buffer.data();
storedValueSize = compressRet;
compressed = 1;
}
#endif
const auto keyHash = XXH128(key, keySize, 0);
check(sqlite3_bind_blob64(store_stmt, 1, &keyHash, sizeof(keyHash), SQLITE_TRANSIENT)); check(sqlite3_bind_blob64(store_stmt, 1, &keyHash, sizeof(keyHash), SQLITE_TRANSIENT));
check( check(sqlite3_bind_blob64(store_stmt, 2, storedValue, storedValueSize, SQLITE_STATIC));
sqlite3_bind_blob64(store_stmt, 2, storedValue, storedValueSize, compressed ? SQLITE_STATIC : SQLITE_TRANSIENT)); check(sqlite3_bind_int64(store_stmt, 3, static_cast<sqlite3_int64>(blob.size())));
check(sqlite3_bind_int64(store_stmt, 3, static_cast<sqlite3_int64>(valueSize)));
check(sqlite3_bind_int(store_stmt, 4, compressed)); check(sqlite3_bind_int(store_stmt, 4, compressed));
check(sqlite3_bind_int64(store_stmt, 5, days_now())); check(sqlite3_bind_int64(store_stmt, 5, days_now()));
g_stores.fetch_add(1, std::memory_order_relaxed);
const auto ret = sqlite3_step(store_stmt); const auto ret = sqlite3_step(store_stmt);
check(sqlite3_reset(store_stmt));
check(sqlite3_bind_null(store_stmt, 2));
if (ret != SQLITE_DONE) { if (ret != SQLITE_DONE) {
// Error or something
Log.error("Failed to insert row: {}", sqlite3_errmsg(db)); Log.error("Failed to insert row: {}", sqlite3_errmsg(db));
return; return;
} }
check(sqlite3_reset(store_stmt));
check(sqlite3_bind_null(store_stmt, 2));
check(sqlite3_bind_null(store_stmt, 4));
tx.commit(); tx.commit();
} }
static void cache_writer_loop() {
std::vector<uint8_t> compressBuffer;
for (;;) {
XXH128_hash_t keyHash{};
QueuedBlob blob;
{
std::unique_lock lock(g_writeMutex);
g_writeCv.wait(lock, [] { return g_writerStop || !g_writeOrder.empty(); });
if (g_writeOrder.empty()) {
// Stopping, and everything queued has been written.
break;
}
keyHash = g_writeOrder.front();
g_writeOrder.pop_front();
blob = g_queuedBlobs.at(keyHash);
g_writerBusy = true;
}
const auto start = std::chrono::steady_clock::now();
write_blob(keyHash, *blob, compressBuffer);
if (blob->size() >= LargeBlobBytes) {
const auto elapsed =
std::chrono::duration_cast<std::chrono::milliseconds>(std::chrono::steady_clock::now() - start);
Log.info("Wrote a {:.1f} MiB Dawn cache blob in {} ms, off the GPU device",
static_cast<double>(blob->size()) / (1024.0 * 1024.0), elapsed.count());
}
{
std::lock_guard lock(g_writeMutex);
if (const auto entry = g_queuedBlobs.find(keyHash); entry != g_queuedBlobs.end()) {
if (entry->second == blob) {
g_queuedBlobs.erase(entry);
} else {
// Stored again while this copy was being written: the newer content still goes out.
g_writeOrder.push_back(keyHash);
}
}
g_writerBusy = false;
}
g_writeCv.notify_all();
}
}
void store_to_cache(void const* key, size_t keySize, void const* value, size_t valueSize, void*) {
const auto keyHash = XXH128(key, keySize, 0);
const auto* bytes = static_cast<const uint8_t*>(value);
const bool large = valueSize >= LargeBlobBytes;
const XXH128_hash_t content = large ? XXH3_128bits(value, valueSize) : XXH128_hash_t{};
if (large) {
std::lock_guard lock(g_writeMutex);
if (const auto known = g_largeContent.find(keyHash);
known != g_largeContent.end() && XXH128_isEqual(known->second, content)) {
g_unchanged.fetch_add(1, std::memory_order_relaxed);
return;
}
}
// Dawn's buffer is only valid for this call.
auto blob = std::make_shared<const std::vector<uint8_t>>(bytes, bytes + valueSize);
{
std::lock_guard lock(g_writeMutex);
if (large) {
g_largeContent.insert_or_assign(keyHash, content);
}
if (g_queuedBlobs.insert_or_assign(keyHash, std::move(blob)).second) {
g_writeOrder.push_back(keyHash);
}
if (!g_writerThread.joinable()) {
g_writerStop = false;
g_writerThread = std::thread(cache_writer_loop);
}
}
g_stores.fetch_add(1, std::memory_order_relaxed);
g_writeCv.notify_all();
}
void flush_cache_writes() {
std::unique_lock lock(g_writeMutex);
g_writeCv.wait(lock, [] { return g_writeOrder.empty() && !g_writerBusy; });
}
void cache_shutdown() { void cache_shutdown() {
{
std::lock_guard lock(g_writeMutex);
g_writerStop = true;
}
g_writeCv.notify_all();
// The writer drains the queue before it exits.
if (g_writerThread.joinable()) {
g_writerThread.join();
}
{
std::lock_guard lock(g_writeMutex);
g_queuedBlobs.clear();
g_writeOrder.clear();
g_largeContent.clear();
g_writerStop = false;
}
std::lock_guard lock(cache_mutex); std::lock_guard lock(cache_mutex);
#if defined(AURORA_CACHE_USE_ZSTD)
compress_buffer.clear();
#endif
flush_touches(); flush_touches();
check(sqlite3_finalize(load_stmt)); check(sqlite3_finalize(load_stmt));
check(sqlite3_finalize(store_stmt)); check(sqlite3_finalize(store_stmt));
@@ -420,6 +579,7 @@ BlobCacheStats blob_cache_stats() noexcept {
.hits = g_hits.load(std::memory_order_relaxed), .hits = g_hits.load(std::memory_order_relaxed),
.stores = g_stores.load(std::memory_order_relaxed), .stores = g_stores.load(std::memory_order_relaxed),
.hitBytes = g_hitBytes.load(std::memory_order_relaxed), .hitBytes = g_hitBytes.load(std::memory_order_relaxed),
.unchanged = g_unchanged.load(std::memory_order_relaxed),
}; };
} }
+14
View File
@@ -106,6 +106,20 @@ if (AURORA_ENABLE_GX)
) )
gtest_discover_tests(gx_fifo_tests) gtest_discover_tests(gx_fifo_tests)
# The Dawn blob cache's background writer, against a real SQLite database in a temp folder.
if (TARGET sqlite3)
add_executable(gpu_cache_tests gpu_cache_test.cpp ../lib/webgpu/gpu_cache.cpp)
target_include_directories(gpu_cache_tests PRIVATE ../include ../lib)
target_compile_definitions(gpu_cache_tests PRIVATE AURORA TARGET_PC)
target_link_libraries(gpu_cache_tests PRIVATE gtest gtest_main fmt::fmt xxhash sqlite3 dawn::dawncpp_headers
TracyClient ${AURORA_SDL3_TARGET})
if (AURORA_CACHE_USE_ZSTD)
target_compile_definitions(gpu_cache_tests PRIVATE AURORA_CACHE_USE_ZSTD)
target_link_libraries(gpu_cache_tests PRIVATE libzstd_static)
endif ()
gtest_discover_tests(gpu_cache_tests)
endif ()
endif () # AURORA_ENABLE_GX endif () # AURORA_ENABLE_GX
# DVD API tests # DVD API tests
+159
View File
@@ -0,0 +1,159 @@
// The Dawn blob cache writes off Dawn's device lock: store_to_cache copies and queues, a writer
// thread compresses and commits. These tests pin what callers rely on: a stored blob reads back
// at once, reaches the database, is not rewritten when unchanged, and survives shutdown.
#include "webgpu/gpu.hpp"
#include "internal.hpp"
#include <gtest/gtest.h>
#include <cstdint>
#include <cstdio>
#include <filesystem>
#include <string>
#include <vector>
namespace aurora {
AuroraConfig g_config{};
void log_internal(AuroraLogLevel, const char*, const char* message, unsigned int len) noexcept {
std::fprintf(stderr, "%.*s\n", static_cast<int>(len), message);
}
void Module::show_fatal_dialog(const char*, std::string_view) noexcept {}
} // namespace aurora
auto fmt::formatter<AuroraLogLevel>::format(AuroraLogLevel level, format_context& ctx) const
-> format_context::iterator {
return fmt::format_to(ctx.out(), "{}", static_cast<int>(level));
}
namespace {
using aurora::webgpu::blob_cache_stats;
using aurora::webgpu::cache_shutdown;
using aurora::webgpu::flush_cache_writes;
using aurora::webgpu::load_from_cache;
using aurora::webgpu::store_to_cache;
class GpuCacheTest : public ::testing::Test {
protected:
void SetUp() override {
m_directory = std::filesystem::temp_directory_path() /
("aurora_gpu_cache_test_" + std::string(::testing::UnitTest::GetInstance()->current_test_info()->name()));
std::filesystem::remove_all(m_directory);
std::filesystem::create_directories(m_directory);
m_path = m_directory.string();
aurora::g_config.cachePath = m_path.c_str();
}
void TearDown() override {
cache_shutdown();
std::filesystem::remove_all(m_directory);
}
static void Store(const std::string& key, const std::vector<uint8_t>& value) {
store_to_cache(key.data(), key.size(), value.data(), value.size(), nullptr);
}
// Dawn's protocol: probe for the size, then fetch into a buffer of that size.
static std::vector<uint8_t> Load(const std::string& key) {
const size_t size = load_from_cache(key.data(), key.size(), nullptr, 0, nullptr);
std::vector<uint8_t> value(size);
if (size != 0) {
EXPECT_EQ(load_from_cache(key.data(), key.size(), value.data(), value.size(), nullptr), size);
}
return value;
}
// Pseudo-random bytes, so compression cannot hide a mix-up between two blobs.
static std::vector<uint8_t> Bytes(size_t size, uint32_t seed) {
std::vector<uint8_t> value(size);
uint32_t state = seed * 2654435761u + 1u;
for (auto& byte : value) {
state = state * 1664525u + 1013904223u;
byte = static_cast<uint8_t>(state >> 24);
}
return value;
}
std::filesystem::path m_directory;
std::string m_path;
};
TEST_F(GpuCacheTest, StoredBlobReadsBackBeforeAndAfterItIsCommitted) {
const auto value = Bytes(4096, 1);
Store("pipeline", value);
EXPECT_EQ(Load("pipeline"), value);
flush_cache_writes();
EXPECT_EQ(Load("pipeline"), value);
EXPECT_TRUE(Load("missing").empty());
}
TEST_F(GpuCacheTest, ShutdownWritesWhatIsStillQueued) {
const auto value = Bytes(3 << 20, 2);
Store("monolithic", value);
cache_shutdown();
// The next call reopens the database: only a committed row can answer now.
EXPECT_EQ(Load("monolithic"), value);
}
TEST_F(GpuCacheTest, NewestStoreOfAKeyWins) {
const auto first = Bytes(2 << 20, 3);
const auto second = Bytes(2 << 20, 4);
Store("monolithic", first);
Store("monolithic", second);
EXPECT_EQ(Load("monolithic"), second);
flush_cache_writes();
cache_shutdown();
EXPECT_EQ(Load("monolithic"), second);
}
TEST_F(GpuCacheTest, UnchangedLargeBlobIsNotRewritten) {
const auto value = Bytes(2 << 20, 5);
Store("monolithic", value);
flush_cache_writes();
const auto before = blob_cache_stats();
Store("monolithic", value);
auto after = blob_cache_stats();
EXPECT_EQ(after.stores, before.stores);
EXPECT_EQ(after.unchanged, before.unchanged + 1);
auto changed = value;
changed[changed.size() / 2] ^= 0xFF;
Store("monolithic", changed);
after = blob_cache_stats();
EXPECT_EQ(after.stores, before.stores + 1);
EXPECT_EQ(Load("monolithic"), changed);
// Changing back is a change too.
Store("monolithic", value);
EXPECT_EQ(blob_cache_stats().stores, before.stores + 2);
EXPECT_EQ(Load("monolithic"), value);
}
TEST_F(GpuCacheTest, BlobLoadedAtStartupIsNotRewrittenUnchanged) {
const auto value = Bytes(2 << 20, 6);
Store("monolithic", value);
cache_shutdown();
// A new session loads the blob Dawn saved last time, then serializes the same bytes again.
ASSERT_EQ(Load("monolithic"), value);
const auto before = blob_cache_stats();
Store("monolithic", value);
const auto after = blob_cache_stats();
EXPECT_EQ(after.stores, before.stores);
EXPECT_EQ(after.unchanged, before.unchanged + 1);
}
TEST_F(GpuCacheTest, SmallBlobsAreAlwaysWritten) {
const auto value = Bytes(512, 7);
Store("shader", value);
flush_cache_writes();
const auto before = blob_cache_stats();
Store("shader", value);
EXPECT_EQ(blob_cache_stats().stores, before.stores + 1);
flush_cache_writes();
EXPECT_EQ(Load("shader"), value);
}
} // namespace
+1
View File
@@ -603,6 +603,7 @@ Bring-up fixes that only a device could reveal:
| Crypto++ `cpu-features.h` not found | The NDK ships cpu-features as source | Compiled into `mkw_cryptopp` on Android | | Crypto++ `cpu-features.h` not found | The NDK ships cpu-features as source | Compiled into `mkw_cryptopp` on Android |
| Exploded racers and menu characters; smeared movie panels in the menus; then, once those were fixed, damaged eyes and slightly misplaced detail on characters | The Adreno 740 driver reads the wrong bytes when the shader multiplies an index by a stride that is not a multiple of 4. That covers the vertex fetch (`ubuf.vtx_start + vidx * stride + offset`) and indexed array reads (`array_start + index * stride`, e.g. 6-byte S16 normals). GX packs both byte-tight, so skinned models (a 1-byte `PNMTXIDX` first, stride 7) broke everywhere | Android pads every uploaded vertex and every indexed-array element to a 4-byte stride (`padded_upload_stride` in `lib/gx/gx.cpp`). Offsets inside a vertex or element are unchanged, and desktop is unchanged. **Fixed, headset-verified 2026-09-16** at character select and a Grand Prix start | | Exploded racers and menu characters; smeared movie panels in the menus; then, once those were fixed, damaged eyes and slightly misplaced detail on characters | The Adreno 740 driver reads the wrong bytes when the shader multiplies an index by a stride that is not a multiple of 4. That covers the vertex fetch (`ubuf.vtx_start + vidx * stride + offset`) and indexed array reads (`array_start + index * stride`, e.g. 6-byte S16 normals). GX packs both byte-tight, so skinned models (a 1-byte `PNMTXIDX` first, stride 7) broke everywhere | Android pads every uploaded vertex and every indexed-array element to a 4-byte stride (`padded_upload_stride` in `lib/gx/gx.cpp`). Offsets inside a vertex or element are unchanged, and desktop is unchanged. **Fixed, headset-verified 2026-09-16** at character select and a Grand Prix start |
| Every launch recompiled every shader: a 14 to 34 s prewarm, and the Dawn blob cache reporting exactly one miss and no stores | Dawn's monolithic Vulkan pipeline cache is only written by `PerformIdleTasks`, which `gpu.cpp` resolved through the Windows DLL alone, and the quit path ends the process without `aurora_shutdown`, so nothing compiled after prewarm was kept either | Static Dawn calls it directly; `aurora_store_pipeline_caches` runs at a race exit, when the session loses focus and on the quit path, and the compiler stores idle bursts itself while the headset shows the virtual screen. The unpacked `initial_pipeline_cache.db` is also refreshed per APK install now | | Every launch recompiled every shader: a 14 to 34 s prewarm, and the Dawn blob cache reporting exactly one miss and no stores | Dawn's monolithic Vulkan pipeline cache is only written by `PerformIdleTasks`, which `gpu.cpp` resolved through the Windows DLL alone, and the quit path ends the process without `aurora_shutdown`, so nothing compiled after prewarm was kept either | Static Dawn calls it directly; `aurora_store_pipeline_caches` runs at a race exit, when the session loses focus and on the quit path, and the compiler stores idle bursts itself while the headset shows the virtual screen. The unpacked `initial_pipeline_cache.db` is also refreshed per APK install now |
| The game and its music froze for 0.9 to 3.8 s at the end of a race, longer as a session went on | Dawn serializes its monolithic Vulkan pipeline cache (58 MiB on the Quest, and it barely compresses) under its device lock and hands it to Aurora's blob-cache callback, which compressed it and committed it to SQLite still inside that lock; the frame worker waited on the device and the game thread on the frame worker. The store also ran on the XR pacing thread | `gpu_cache.cpp` copies the blob and queues it; a writer thread compresses and commits, and bytes identical to what the database holds are not rewritten. The race exit uses `aurora_request_pipeline_cache_store`, off the pacing thread. The log's `Stored the pipeline caches` line reports the part still spent holding the device |
How the explosion was isolated, so the next Adreno rendering bug starts further How the explosion was isolated, so the next Adreno rendering bug starts further
ahead: ahead:
+7 -4
View File
@@ -785,6 +785,7 @@ private:
ResetTrackingOrigin(); ResetTrackingOrigin();
// Nothing is displayed while the session is not running (the system menu, // Nothing is displayed while the session is not running (the system menu,
// the headset taken off), so the stall of a cache store is invisible here. // the headset taken off), so the stall of a cache store is invisible here.
// Waiting for it is deliberate: the process may be ended next.
aurora_store_pipeline_caches(); aurora_store_pipeline_caches();
} }
} }
@@ -851,9 +852,11 @@ private:
aurora_set_stereo_panel_layer(panel_layer); aurora_set_stereo_panel_layer(panel_layer);
presentation.panel.requested = panel_layer && OpenXRSettingsPanelOpen(); presentation.panel.requested = panel_layer && OpenXRSettingsPanelOpen();
// Pipeline caches are stored where their stall is invisible: once when a race ends, // Pipeline caches are stored where their stall is least visible: once when a race
// and by the compiler itself while the headset shows the virtual screen. Never // ends, and by the compiler itself while the headset shows the virtual screen. Never
// mid-race, and a race on the virtual screen (Flat Screen mode) is still a race. // mid-race, and a race on the virtual screen (Flat Screen mode) is still a race. The
// race-exit store runs on Aurora's thread: this one keeps submitting frames while
// Dawn holds its device to serialize, which is still a brief game stall.
const bool racing = immersive || (!policy.config.immersive_races && const bool racing = immersive || (!policy.config.immersive_races &&
policy.scene.mode == VRSceneMode::Race); policy.scene.mode == VRSceneMode::Race);
if (!store_gate_set || racing != store_gate_racing) { if (!store_gate_set || racing != store_gate_racing) {
@@ -862,7 +865,7 @@ private:
store_gate_racing = racing; store_gate_racing = racing;
aurora_set_pipeline_cache_idle_store(!racing); aurora_set_pipeline_cache_idle_store(!racing);
if (left_race) { if (left_race) {
aurora_store_pipeline_caches(); aurora_request_pipeline_cache_store();
} }
} }