Quest Fixed Vulkan pipeline cache management and storage mechanisms

This commit is contained in:
iChris4 committed 2026-09-18 23:38:55 +02:00
1 parent bf3b53555b
commit d4a3f0b350
10 files changed
+176 -19

No files matched your search

+7
View File
@@ -254,6 +254,13 @@ bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros);
// worker is fully done. Unlike aurora_wait_for_frame_worker(), this is safe in
// the gap between aurora_end_frame() and aurora_begin_frame().
void aurora_quiesce_frame_worker();
// Persists the renderer's pipeline caches now (Dawn's Vulkan pipeline cache, then the queued
// pipeline recipes). It holds the GPU device for the store, so a race would see it as a stall:
// call it at a race exit, when the session loses focus, or before ending the process.
void aurora_store_pipeline_caches();
// Allows the pipeline compiler to store the caches itself, rate-limited, whenever a first-use
// burst completes. Off by default; enable it while a stall is acceptable, such as in menus.
void aurora_set_pipeline_cache_idle_store(bool allowed);
// Absolute schedule for the next sealed frame, on steady_clock: baseNanos anchors the group and
// intervalNanos is the period, so slot k of N+1 fires at base + k*interval/(N+1). Zeros clear it.
void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos);
+15
View File
@@ -24,6 +24,9 @@
#include <magic_enum.hpp>
#include "android_debug.hpp"
#ifdef AURORA_ENABLE_GX
#include "gfx/pipeline_cache.hpp"
#endif
#include "system_info.hpp"
#include "tracy/Tracy.hpp"
@@ -2509,6 +2512,18 @@ bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros) {
return aurora::wait_for_frame_worker_for(std::chrono::microseconds(timeoutMicros));
}
void aurora_quiesce_frame_worker() { aurora::quiesce_frame_worker(); }
void aurora_store_pipeline_caches() {
#ifdef AURORA_ENABLE_GX
aurora::gfx::store_pipeline_caches();
#endif
}
void aurora_set_pipeline_cache_idle_store(bool allowed) {
#ifdef AURORA_ENABLE_GX
aurora::gfx::set_pipeline_cache_idle_store(allowed);
#else
(void)allowed;
#endif
}
void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos) {
aurora::g_presentScheduleBaseNanos.store(baseNanos, std::memory_order_release);
aurora::g_presentScheduleIntervalNanos.store(intervalNanos, std::memory_order_release);
+98 -12
View File
@@ -93,6 +93,11 @@ static std::mutex g_pipelineCacheWriterMutex;
static std::deque<PipelineCacheWrite> g_pipelineCacheWriteQueue;
static absl::flat_hash_set<PipelineRef> g_pipelineCachePendingWrites;
static bool g_pipelineCacheWriterStop = false;
// Set while the writer thread is alive and while it is inside a write transaction, both under
// g_pipelineCacheWriterMutex, so a flush can wait for the queue to reach the database.
static bool g_pipelineCacheWriterRunning = false;
static bool g_pipelineCacheWriterBusy = false;
static std::condition_variable g_pipelineCacheWriterIdleCv;
static int g_sdlVfsRegisterResult = SQLITE_ERROR;
static SdlVfsSqliteFile* sdl_vfs_file(sqlite3_file* file) {
@@ -909,10 +914,13 @@ static void pipeline_cache_writer() {
g_pipelineCacheWriterCv.wait(lock,
[] { return g_pipelineCacheWriterStop || !g_pipelineCacheWriteQueue.empty(); });
if (g_pipelineCacheWriterStop && g_pipelineCacheWriteQueue.empty()) {
g_pipelineCacheWriterRunning = false;
g_pipelineCacheWriterIdleCv.notify_all();
return;
}
batch.swap(g_pipelineCacheWriteQueue);
g_pipelineCachePendingWrites.clear();
g_pipelineCacheWriterBusy = true;
}
bool writeFailed = false;
@@ -935,6 +943,15 @@ static void pipeline_cache_writer() {
}
}
{
std::lock_guard lock{g_pipelineCacheWriterMutex};
g_pipelineCacheWriterBusy = false;
if (writeFailed) {
g_pipelineCacheWriterRunning = false;
}
g_pipelineCacheWriterIdleCv.notify_all();
}
if (writeFailed) {
pipeline_cache_abort();
return;
@@ -948,21 +965,48 @@ static std::atomic_bool g_prewarmActive{false};
static std::chrono::steady_clock::time_point g_prewarmStart{};
static uint32_t g_prewarmCount = 0;
// Storing the monolithic Vulkan pipeline cache holds the device lock for the whole
// vkGetPipelineCacheData, compression and database write, a visible stall mid-race. Outside boot
// prewarm it therefore runs only at moments the host declares safe: on request, through
// store_pipeline_caches at a race exit and before the process ends, and, while the host allows
// idle stores in menus, when a first-use burst drains, at most once per interval. Dawn skips the
// store when no pipeline was created since the last one, but a creation served from the cache
// still counts, so every store rewrites the whole blob; the interval keeps that rare.
static std::mutex g_storeMutex;
static std::atomic_bool g_idleStoreAllowed{false};
static std::chrono::steady_clock::time_point g_lastStore{};
constexpr auto kIdleStoreInterval = std::chrono::seconds(120);
static void note_pipeline_queue_drained() {
if (!g_prewarmActive.exchange(false, std::memory_order_acq_rel)) {
if (g_prewarmActive.exchange(false, std::memory_order_acq_rel)) {
{
std::lock_guard lock{g_storeMutex};
webgpu::serialize_pipeline_caches();
g_lastStore = std::chrono::steady_clock::now();
}
const auto elapsed =
std::chrono::duration_cast<std::chrono::milliseconds>(std::chrono::steady_clock::now() - g_prewarmStart);
const auto stats = webgpu::blob_cache_stats();
Log.info("Pipeline prewarm finished: {} pipelines in {:.1f} s (Dawn blob cache: {}/{} hits, {} stores, {:.1f} MiB "
"loaded)",
g_prewarmCount, elapsed.count() / 1000.0, stats.hits, stats.lookups, stats.stores,
static_cast<double>(stats.hitBytes) / (1024.0 * 1024.0));
return;
}
// Persist the monolithic Vulkan pipeline cache once after boot prewarm only; doing it on
// every drained burst stalls the device lock mid-race. Later first-use compiles are
// covered by the shutdown serialize.
if (!g_idleStoreAllowed.load(std::memory_order_relaxed)) {
return;
}
// A store in progress on another worker covers this burst too.
std::unique_lock lock{g_storeMutex, std::try_to_lock};
if (!lock.owns_lock()) {
return;
}
const auto now = std::chrono::steady_clock::now();
if (now - g_lastStore < kIdleStoreInterval) {
return;
}
g_lastStore = now;
webgpu::serialize_pipeline_caches();
const auto elapsed =
std::chrono::duration_cast<std::chrono::milliseconds>(std::chrono::steady_clock::now() - g_prewarmStart);
const auto stats = webgpu::blob_cache_stats();
Log.info("Pipeline prewarm finished: {} pipelines in {:.1f} s (Dawn blob cache: {}/{} hits, {} stores, {:.1f} MiB "
"loaded)",
g_prewarmCount, elapsed.count() / 1000.0, stats.hits, stats.lookups, stats.stores,
static_cast<double>(stats.hitBytes) / (1024.0 * 1024.0));
}
static void compile_pending_pipeline(PendingPipeline pending) {
@@ -1130,10 +1174,28 @@ static void start_pipeline_cache_writer() {
return;
}
g_pipelineCacheWriterStop = false;
{
std::lock_guard lock{g_pipelineCacheWriterMutex};
g_pipelineCacheWriterStop = false;
g_pipelineCacheWriterRunning = true;
g_pipelineCacheWriterBusy = false;
}
g_pipelineCacheWriterThread = std::thread(pipeline_cache_writer);
}
// Waits, bounded, until every queued recipe row has reached the database: the caller is about to
// end the process, or wants the store it just made to be complete on disk.
static void flush_pipeline_cache_writes() {
std::unique_lock lock{g_pipelineCacheWriterMutex};
if (!g_pipelineCacheWriterRunning) {
return;
}
g_pipelineCacheWriterCv.notify_one();
g_pipelineCacheWriterIdleCv.wait_for(lock, std::chrono::seconds(2), [] {
return !g_pipelineCacheWriterRunning || (g_pipelineCacheWriteQueue.empty() && !g_pipelineCacheWriterBusy);
});
}
static void stop_pipeline_cache_writer() {
if (g_pipelineCacheWriterThread.joinable()) {
{
@@ -1146,10 +1208,32 @@ static void stop_pipeline_cache_writer() {
g_pipelineCacheWriterStop = false;
}
g_pipelineCacheWriterRunning = false;
g_pipelineCacheWriterBusy = false;
g_pipelineCacheWriteQueue.clear();
g_pipelineCachePendingWrites.clear();
}
void store_pipeline_caches() {
const uint64_t storesBefore = webgpu::blob_cache_stats().stores;
const auto start = std::chrono::steady_clock::now();
{
std::lock_guard lock{g_storeMutex};
webgpu::serialize_pipeline_caches();
g_lastStore = std::chrono::steady_clock::now();
}
flush_pipeline_cache_writes();
const uint64_t stored = webgpu::blob_cache_stats().stores - storesBefore;
if (stored != 0) {
const auto elapsed = std::chrono::duration_cast<std::chrono::milliseconds>(std::chrono::steady_clock::now() - start);
Log.info("Stored the pipeline caches in {} ms: {} new Dawn blob cache entries", elapsed.count(), stored);
}
}
void set_pipeline_cache_idle_store(bool allowed) noexcept {
g_idleStoreAllowed.store(allowed, std::memory_order_relaxed);
}
template <>
PipelineRef find_pipeline(ShaderType type, const clear::PipelineConfig& config, NewPipelineCallback&& cb) {
return find_pipeline_impl(type, config, std::move(cb), true, std::nullopt);
@@ -1163,6 +1247,8 @@ PipelineRef find_pipeline(ShaderType type, const gx::PipelineConfig& config, New
void initialize_pipeline_cache() {
g_pipelineCacheBroken = false;
g_pipelineCacheWriterStop = false;
g_idleStoreAllowed.store(false, std::memory_order_relaxed);
g_lastStore = {};
g_pipelineFrameActive = false;
g_pipelineThreadEnd = false;
g_activeBackgroundPipelineWorkers = 0;
+7
View File
@@ -23,6 +23,13 @@ void end_pipeline_frame();
void set_skip_unready_pipelines(bool enabled) noexcept;
bool skip_unready_pipelines() noexcept;
uint32_t queued_pipeline_count() noexcept;
// Persists the pipeline caches now: Dawn's Vulkan pipeline cache when a pipeline was compiled
// since the last store, then every queued recipe row. Holds the device for the store, so call it
// where a stall is invisible (a race exit, process exit).
void store_pipeline_caches();
// Lets a drained first-use burst store the caches itself, rate-limited. Off by default; a host
// turns it on while a stall is acceptable (menus) and off again for a race.
void set_pipeline_cache_idle_store(bool allowed) noexcept;
template <typename Config>
PipelineRef find_pipeline(ShaderType type, const Config& config, NewPipelineCallback&& cb);
+11 -1
View File
@@ -913,10 +913,16 @@ void fail_if_device_lost() noexcept {
}
void serialize_pipeline_caches() noexcept {
#if defined(WEBGPU_DAWN) && defined(_WIN32)
#if defined(WEBGPU_DAWN)
// Only the Vulkan backend keeps a monolithic VkPipelineCache (toggle above). Dawn writes it to
// the blob cache from PerformIdleTasks, and only when a pipeline was compiled since the last
// store, so calling this when nothing changed is cheap.
if (!g_device || g_backendType != wgpu::BackendType::Vulkan) {
return;
}
#if defined(_WIN32)
// The Windows product links the Dawn DLL from llvm-mingw, which cannot call the exported C++
// symbol directly; resolve its MSVC-mangled name instead.
using PerformIdleTasksFn = void(*)(const wgpu::Device*);
static const auto performIdleTasks = []() -> PerformIdleTasksFn {
const HMODULE dawnModule = GetModuleHandleW(L"webgpu_dawn.dll");
@@ -929,6 +935,10 @@ void serialize_pipeline_caches() noexcept {
if (performIdleTasks != nullptr) {
performIdleTasks(&g_device);
}
#elif !defined(__MINGW32__)
// Static Dawn (Android, Linux): DawnNative.h is included above.
dawn::native::PerformIdleTasks(g_device);
#endif
#endif
}